fapony 0.1.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +75 -44
  2. package/fapony.ts +32 -1
  3. package/package.json +3 -2
  4. package/skill/move-to-done/SKILL.md +18 -28
  5. package/skill/plan-with-pony/SKILL.md +8 -3
  6. package/src/analyze.ts +13 -1
  7. package/src/conventions-seed.ts +0 -1
  8. package/src/db/defaults.ts +1 -1
  9. package/src/db/getters.ts +3 -3
  10. package/src/db/store.ts +0 -75
  11. package/src/db/types.ts +1 -1
  12. package/src/debt.ts +169 -34
  13. package/src/digest/collect.ts +16 -0
  14. package/src/digest/text.ts +25 -0
  15. package/src/hook.ts +424 -16
  16. package/src/init-mem.ts +123 -37
  17. package/src/init.ts +51 -34
  18. package/src/install/claude.ts +7 -5
  19. package/src/install/opencode.ts +59 -5
  20. package/src/lint-baseline.ts +1 -7
  21. package/src/mcp/evidence.ts +1 -1
  22. package/src/mcp/tools/index.ts +97 -185
  23. package/src/mcp/tools/mem.ts +153 -11
  24. package/src/mcp/transport.ts +5 -13
  25. package/src/mem/commands/read.ts +448 -0
  26. package/src/mem/commands/where.ts +56 -0
  27. package/{templates → src}/mem/commands/write.ts +12 -5
  28. package/src/mem/index.ts +144 -0
  29. package/src/mem/store.ts +348 -0
  30. package/src/memory.ts +237 -40
  31. package/src/plan-seed.ts +20 -1
  32. package/src/review-seed.ts +19 -0
  33. package/src/setup.ts +4 -6
  34. package/src/stats/data.ts +34 -86
  35. package/src/stats/format.ts +8 -9
  36. package/src/stats/index.ts +0 -1
  37. package/templates/PLAN.md +1 -0
  38. package/src/mcp/tools/context.ts +0 -66
  39. package/src/mcp/tools/plans.ts +0 -255
  40. package/src/mcp/tools/stats.ts +0 -96
  41. package/templates/mem/commands/read.ts +0 -194
  42. package/templates/mem/commands/selftest.ts +0 -450
  43. package/templates/mem/mem.ts +0 -68
  44. package/templates/mem/store.ts +0 -285
  45. /package/{templates → src}/mem/commands/plan.ts +0 -0
  46. /package/{templates → src}/mem/commands/rotate.ts +0 -0
  47. /package/{templates → src}/mem/render.ts +0 -0
  48. /package/{templates → src}/mem/selectors.ts +0 -0
package/README.md CHANGED
@@ -37,7 +37,7 @@ quietly counted as free.
37
37
  </details>
38
38
 
39
39
  That is day one. Past that, fapony measures what coding agents actually do — rounds, pass/fail,
40
- cost per grade — through 6 MCP tools any agent can call. If you juggle more than one agent, this is
40
+ cost per grade — through 4 MCP tools any agent can call. If you juggle more than one agent, this is
41
41
  the point: the numbers come from the same yardstick everywhere, so "which model earns its keep on
42
42
  which kind of task" becomes a data question instead of a vibe. On top of measurement it checks
43
43
  claims against git facts: handoff conformance, allowlisted evidence, a 6-grade verdict — with
@@ -58,7 +58,19 @@ on you to hold: work isn't randomly assigned to models, so a gap this size is a
58
58
  controlled trial — you likely route easy tasks to the cheap model already. `n≥5` is fapony's own
59
59
  floor before a model counts toward the frontier at all; below that it's a data point, not a pick.
60
60
 
61
- **The reason to keep it running is the third layer: knowledge accumulation.** Any single client already logs its own session — timing, tokens, tool calls. What none of them see is *across* runs, clients and task shapes: which model earns its keep on which kind of work **in this project**, at what token cost, graded by whoever reviewed it. Every verdict carries a `regime` (`code` / `fix` / `review` / `plan` / `inquiry` / `test`), and runs split by whether there was a plan at all — so "does planning beat diving in, and for which model" is a table, not an argument.
61
+ **The reason to keep it running is the third layer: knowledge accumulation — and the thing it
62
+ accumulates is pain.** An agent has no memory of pain across sessions: it writes the 37th
63
+ hand-rolled `try/catch` as cheerfully as the first, because every session starts new. Wrappers and
64
+ shared libraries get built by *people* who were hurt by the same thing often enough to remember.
65
+ That is why a codebase written with agents from day one tends not to grow a shared layer — nobody
66
+ in the room remembers. fapony is the part that remembers: graded verdicts and mem rows both carry
67
+ `files[]`, so the zones that keep coming back in failed and re-done work are a query, not a hunch.
68
+ Paired with `fapony debt`, which tracks how far the codebase has actually moved to a convention you
69
+ already decided on, that is the loop: notice the repeated cost, name the shared thing, watch the
70
+ migration finish. Finding dead code and duplication is *not* part of it — knip and friends already
71
+ do that better, and a convention with a `checker` is deliberately left to the checker.
72
+
73
+ **The measurement layer underneath it:** Any single client already logs its own session — timing, tokens, tool calls. What none of them see is *across* runs, clients and task shapes: which model earns its keep on which kind of work **in this project**, at what token cost, graded by whoever reviewed it. Every verdict carries a `regime` (`code` / `fix` / `review` / `plan` / `inquiry` / `test`), and runs split by whether there was a plan at all — so "does planning beat diving in, and for which model" is a table, not an argument.
62
74
 
63
75
  Three tiers, deliberately: **measurement ships today** and needs no per-project setup — raw facts nobody can call unfair. **Verification is the sharper edge** but stays beta until its evidence layer is hardened; fapony doesn't control your agent's flow, so it never promises "verified" as a headline. **Knowledge accumulation is the compounding one** — it's worthless on run 1 and gets more useful every run after, which is exactly why it's the layer competitors can't clone by copying a feature list.
64
76
 
@@ -72,11 +84,12 @@ Stated up front, because the gap between these two things is where most tooling
72
84
  `.fapony/evidence.json`, and never a command an agent proposes. No allowlist, no evidence.
73
85
  - **It does not judge your code.** `verdict_submit` *stores* a verdict; a human or a reviewing
74
86
  agent supplies it. fapony is the ledger, not the judge.
75
- - **`handoff_check` checks conformance, not correctness.** It verifies that what the agent claimed
76
- lines up with git facts and that it declared its uncertainty — not that the code works. Those are
77
- different guarantees and fapony only offers the first.
78
- - **Nothing blocks.** There is no gate, no hook, no CI failure. Forget to call it and you are back
79
- to exactly the workflow you had.
87
+ - **It checks conformance, not correctness.** What it can verify is that a claim lines up with git
88
+ facts and that uncertainty was declared — not that the code works. Those are different
89
+ guarantees and fapony only offers the first.
90
+ - **Almost nothing blocks.** No CI failure, no gate on your own commands. The one exception is the
91
+ Stop hook, once per turn when a commit ends ungraded; the read hint only annotates. Skip the
92
+ install of both and you are back to exactly the workflow you had.
80
93
  - **Model attribution is inferred, not declared.** A gate is attributed to whichever client
81
94
  session was live in that worktree at that moment. When one model writes the code and another
82
95
  reviews and files the verdict, the grade lands on the reviewer. Reports label it `inferred`;
@@ -109,7 +122,7 @@ fapony usage-scan # scan the session logs already on dis
109
122
  fapony price-scan # fetch the OpenRouter price table → ~/.config/fapony/prices.json
110
123
  fapony usage-web # dashboard; re-run the scans to refresh
111
124
  # both scans are manual by design — nothing fetches or re-reads session logs behind your back
112
- # ask your agent: "Run fapony_stats and fapony_usage — what has it cost me, per model?"
125
+ # ask your agent: "Run fapony_usage — what has it cost me, per model?"
113
126
 
114
127
  # 4. Verify (optional, per project) — scaffold the evidence allowlist
115
128
  fapony init /path/to/your-worktree
@@ -118,7 +131,7 @@ fapony init /path/to/your-worktree
118
131
 
119
132
  With `.fapony/evidence.json` in place, any graded run can be replayed as a report. This one is
120
133
  a CLI command, not an MCP tool — the schemas cost every session of every client and no skill
121
- called them (see [The 6 tools](#the-6-tools) below). Grade something first;
134
+ called them (see [The 4 tools](#the-4-tools) below). Grade something first;
122
135
  `verdict_submit` is what creates the run:
123
136
 
124
137
  ```bash
@@ -147,7 +160,7 @@ flowchart LR
147
160
  F --> G[git facts + session logs]
148
161
  G --> S[stats / usage]
149
162
  G --> V[verification report]
150
- G --> P[project_health - optional]
163
+ G --> M[mem log - what was decided here]
151
164
  ```
152
165
 
153
166
  fapony never drives the agent. It sits on two sides of your work that never touch each
@@ -157,7 +170,7 @@ losing a single number.
157
170
 
158
171
  | | The ledger | The work side |
159
172
  |---|---|---|
160
- | What it is | 6 MCP tools + a SQLite ledger | plans, skills, read-only seed commands |
173
+ | What it is | 4 MCP tools + a SQLite ledger | plans, skills, read-only seed commands |
161
174
  | Needs | an MCP client | nothing — or your own tooling instead |
162
175
  | Writes | one graded row per unit of work | nothing |
163
176
  | Skip it and | there is no fapony | fapony still answers every question |
@@ -185,26 +198,32 @@ sequenceDiagram
185
198
  A->>F: fapony report <run-id>
186
199
  F-->>A: git facts + evidence from .fapony/evidence.json, stamped with server_sha
187
200
  end
188
- A->>F: fapony_stats
189
- F->>L: read across every run, client and project
190
- L-->>A: model x regime x quality — which model to pay for this shape
201
+ Note over A,L: `fapony stats` reads it back — CLI, because you ask it, not the agent
191
202
  ```
192
203
 
193
- The Stop hook is the only thing fapony does *to* you — once per turn, when a commit ends
194
- ungraded. It never picks the grade; it cannot see whether the work held up.
204
+ The Stop hook is the only thing fapony *blocks* — once per turn, when a commit ends ungraded.
205
+ It never picks the grade; it cannot see whether the work held up. The Read hook only annotates:
206
+ one factual line when a read is large enough to be cheaper as `review-seed`, or when the same
207
+ file is read again in a session and its mtime has not moved. The read always proceeds, and
208
+ `FAPONY_NO_REREAD_HINT=1` turns the re-read line off.
195
209
 
196
- ### The 6 tools
210
+ ### The 4 tools
197
211
 
198
212
  | Tool | Tier | Purpose |
199
213
  |------|------|---------|
200
- | `plan_list` | discover | Plan files grouped by state — active / blocked / untouched / superseded / trackers — with a progress tally and each one's run history. Not a raw `ls`; see [Plans your agent can answer questions about](#plans-your-agent-can-answer-questions-about) |
201
- | `fapony_stats` | measure | KPIs across runs: by-model (gates, fail rate, quality, tokens), by-grade, planned vs dove-in, regime x model, per-file risk; `group_by: reason_code\|plan\|file` for top-N slices; `mode: verdict` ranks models by quality vs tokens/pass instead of listing raw counts |
202
214
  | `fapony_usage` | measure | Passive usage from OpenCode, ZCode, Claude Code, and Codex sessions (tokens, cost, by-model; `detail:true` adds per-step timing) |
203
215
  | `verdict_submit` | verify | Store a 6-grade verdict (pass-excellent → uncertain) with a required `regime` — the task shape the grade applies to |
204
- | `project_health_context` | recall | Known-patterns block for the files you are about to touch. Useful when a file does have history; measured across real repos, most do not (1-9% of shipped files come back under a `fix:` within two weeks), so it is optional — never a precondition for editing |
205
- | `mem_find` | recall | Search the project's mem log read-only — decisions/bugs/notes keyed by `files[]`, `text`, `kind` (no default filter), `since`. "What was ever decided about this file?" in one call before editing |
206
-
207
- The handoff/report family is CLI-only — the schemas cost every session of every client and no skill called them. `fapony report <run-id>` prints the full report for a run (facts + handoff conformance + evidence + verdict); `fapony report-web [file]` renders it as a static HTML page (overwrites `file` on every call — safe to reuse the same path). Run `bun run overview` for a one-shot shortcut that writes it to `/tmp/fapony-overview.html` and opens it. `fapony usage-scan` scans session logs and writes a cache file; `fapony usage-web [port]` serves a static HTML dashboard from that cache (no live scanning). Run `fapony usage-scan` periodically to keep data fresh.
216
+ | `mem_find` | recall | Search the project's mem log read-only — decisions/bugs/notes matched on the row's `files[]` (text substring for rows written without it), `text`, `kind` (no default filter), `since`. "What was ever decided about this file?" in one call before editing |
217
+ | `mem_add` | recall | Append a mem row (decision/bug/note/next/hold) with `files[]` required and rejected when empty — the write half of `mem_find`, so the row is findable when you next touch that file |
218
+
219
+ **A tool earns its schema by being called mid-task without being asked.** Everything you invoke
220
+ deliberately is a CLI command instead: the schema is paid as input tokens in every session of
221
+ every client whether or not it is used, while a CLI command costs nothing until it runs. That is
222
+ why the handoff/report family is CLI-only, and why `fapony_stats`, `project_health_context` and
223
+ `plan_list` left the MCP surface in 2026-09 (`fapony stats` answers the first, `fapony mem
224
+ kickoff` the third; the second had no caller).
225
+ Cutting is not the goal — spending where it pays back is: `mem_find` and `verdict_submit` keep
226
+ their schemas because nobody is going to type them at the right moment. `fapony report <run-id>` prints the full report for a run (facts + handoff conformance + evidence + verdict); `fapony report-web [file]` renders it as a static HTML page (overwrites `file` on every call — safe to reuse the same path). Run `bun run overview` for a one-shot shortcut that writes it to `/tmp/fapony-overview.html` and opens it. `fapony usage-scan` scans session logs and writes a cache file; `fapony usage-web [port]` serves a static HTML dashboard from that cache (no live scanning). Run `fapony usage-scan` periodically to keep data fresh.
208
227
 
209
228
  Full protocol, adapter examples (bash, Python), and safety rules: [docs/mcp-handcheck.md](https://github.com/kire21b/fapony/blob/main/docs/mcp-handcheck.md).
210
229
 
@@ -354,29 +373,31 @@ blocks: PLAN-export.md # ordering, stated once instead of buried in pros
354
373
  - [ ] chunk 2 — move overdue out
355
374
  ```
356
375
 
357
- Then ask your agent *"what's left, and what's blocked?"* — `plan_list` answers from the
358
- frontmatter and from fapony's own run history, without reading a single 100KB plan body into
359
- context (`format: "markdown"`):
376
+ Then open the next session with `fapony mem kickoff` — it reads the folder and the mem log and
377
+ prints what is next (priority plans, the first unchecked chunk of each, open bugs) without
378
+ reading a single 100KB plan body into context:
360
379
 
361
380
  ```
362
- ## active — in order (2)
363
- - [ ] PLAN-calendar — 1/3 · unblocks PLAN-export
364
- - [ ] PLAN-export — never attempted
365
- ## blocked (1)
366
- - [ ] PLAN-attendance — waiting: PLAN-documents.md
367
- ## untouched (14) · trackers (3)
368
- done: 63 archived
381
+ ## next up
382
+ [1] chunk 2 — move overdue out (PLAN-calendar.md)
383
+ [2] bug #mu8t5qve — money drifts in the month grid…
384
+ → fapony mem close mu8t5qve "<msg>"
385
+ [3] last touched: src/quick/month.tsx, src/lib/money.ts
369
386
  ```
370
387
 
371
- **Plans with no frontmatter still work** — they are grouped by run history alone (attempted =
372
- active, never attempted = untouched), so an existing folder of plans is queryable before anyone
373
- annotates anything. Two details that keep it honest over years:
388
+ **Plans with no frontmatter still work** — the unchecked checkboxes are enough, so an existing
389
+ folder of plans is usable before anyone annotates anything. Two details that keep it honest over
390
+ years:
374
391
 
375
392
  - The progress tally counts checkboxes in the **first `##` section only**, anchored by position
376
393
  rather than by the word "TL;DR" — so it works in any language, and a step list deeper in the
377
394
  file stays detail instead of becoming status.
378
- - **There is no `MASTER.md`.** Every line of the list above is derived from frontmatter and
379
- checkboxes, so it cannot drift; a hand-kept master file always does.
395
+ - **There is no `MASTER.md`.** Every line above is derived from the plan files themselves, so it
396
+ cannot drift; a hand-kept master file always does.
397
+
398
+ `status` / `blocked_by` / `blocks` / `superseded_by` are read by people, not by a tool — the one
399
+ that read them, `plan_list`, was removed in 2026-09 once `mem kickoff` answered the same
400
+ question from the CLI, where a schema costs nothing until it runs.
380
401
 
381
402
  The layout, and why archiving is a plain `git mv`:
382
403
 
@@ -396,7 +417,7 @@ archived one: [examples/](https://github.com/kire21b/fapony/tree/main/examples).
396
417
 
397
418
  ```bash
398
419
  # Verification & reporting
399
- fapony mcp # MCP server (stdio JSON-RPC — 6 tools)
420
+ fapony mcp # MCP server (stdio JSON-RPC — 4 tools)
400
421
  fapony report <run-id> # verification report for a run
401
422
  fapony report-web [file] # static HTML report page
402
423
  fapony usage-scan # scan session logs → cache (incremental, progress bar)
@@ -407,9 +428,19 @@ fapony digest [--since 7d|YYYY-MM-DD] [--format text|html] [--json] [--out FILE]
407
428
  fapony plan-seed <name> [--spec] [--scope <path>]... # write PLAN (+SPEC): frontmatter, 8 empty sections, prior-art list, ledger context; SPEC chunks carry signatures, every section capped — the agent fills the judgment
408
429
  fapony review-seed [--staged|--commit <sha>|--range <a...b>|--files f1,f2,dir|--plan <PLAN.md>] # read-only scope facts for a review (changed files, importers, untested, signatures, plan cross-check)
409
430
 
431
+ # Memory & convention debt
432
+ fapony mem add <kind> "<text>" --files f1,f2 [spec.md] # append a mem row (decision/bug/note/next/hold)
433
+ fapony mem close <id> "<msg>" # close a bug
434
+ fapony mem find "<text>" # substring-search every row
435
+ fapony mem kickoff [<plan.md>] # open a session + a next-up list
436
+ fapony mem where # show the resolved mem dir and which step won
437
+ fapony mem now | done | stale # views
438
+ fapony debt [--id <convention>] [--where <path>] # ไฟล์ไหนยังไม่ย้ายไป convention ที่ประกาศไว้ (live, read-only)
439
+ fapony lint-baseline [--cmd ...] [--diff] # separate "already red" from "I made it red"
440
+
410
441
  # Setup & maintenance
411
442
  fapony init <path> # scaffold .fapony/ (plan/spec/memory/evidence)
412
- fapony init-mem [--update] # refresh the memory scaffold from the template
443
+ fapony init-mem # delete .memory/ + warn call sites still referencing it
413
444
  fapony install # detect installed clients, prompt to wire each
414
445
  fapony install --all # wire all detected clients without prompting
415
446
  fapony install --platform <name> # force a specific client (bypasses detection)
@@ -426,16 +457,16 @@ fapony test # self-check
426
457
 
427
458
  - `worktrees` — name → absolute path mapping
428
459
  - `review.maxRounds` — round cap enforced by the gate
429
- - `memory` — shell commands for claim/close/add/kickoff, or `null` to default-wire when `.fapony/.memory/mem.ts` exists
430
- - `paths` (`planDir`/`doneDir`/`specDir`/`memoryEntry`/`stateDir`) / `safety` — directory layout and the dangerous-command deny-list
460
+ - `memory` — shell commands for claim/close/add/kickoff, or `null` to default-wire when a `.fapony/.memory/` dir exists
461
+ - `paths` (`planDir`/`doneDir`/`specDir`/`memDir`/`stateDir`) / `safety` — directory layout and the dangerous-command deny-list
431
462
  - `usageWeb` — optional `{ port, hostname }` for `fapony usage-web` server defaults. Run `fapony usage-scan` first to populate the cache.
432
463
 
433
- Env overrides: `FAPONY_CONFIG` (config file), `FAPONY_STATE_DIR` (state DB location; default `~/.config/fapony/`). Full schema, design decisions, and edge cases live with the code in the repo — this README intentionally doesn't duplicate them.
464
+ Env overrides: `FAPONY_CONFIG` (config file), `FAPONY_STATE_DIR` (state DB location; default `~/.config/fapony/`), `FAPONY_NO_REREAD_HINT=1` (turn the re-read hint off). Full schema, design decisions, and edge cases live with the code in the repo — this README intentionally doesn't duplicate them.
434
465
 
435
466
  ## Scope
436
467
 
437
468
  **Supported:**
438
- - MCP server — 6 tools via stdio JSON-RPC, works with any MCP client
469
+ - MCP server — 4 tools via stdio JSON-RPC, works with any MCP client
439
470
  - Measurement: cross-run KPIs by model/grade/value, per-file risk (graded touches vs. fails) + passive usage (tokens, cost)
440
471
  - Model attribution across clients — resolved from the session log that was live when the verdict landed, so a verdict carries a model without the caller declaring one
441
472
  - Zero setup beyond install: the two habits fapony depends on ship in the MCP `initialize` response, not in your rules file
package/fapony.ts CHANGED
@@ -3,6 +3,7 @@
3
3
  // fapony — measure/verify MCP server for coding agents
4
4
  // CLI dispatch: all logic lives in src/
5
5
 
6
+ import { existsSync } from "node:fs";
6
7
  import { cmdAnalyze } from "./src/analyze.js";
7
8
  import { cmdDebt } from "./src/debt.js";
8
9
  import { cmdDigest } from "./src/digest/cli.js";
@@ -12,6 +13,8 @@ import { cmdInitMem } from "./src/init-mem.js";
12
13
  import { cmdInstall } from "./src/install.js";
13
14
  import { cmdLintBaseline } from "./src/lint-baseline.js";
14
15
  import { cmdMcp } from "./src/mcp/transport.js";
16
+ import { cmdMem } from "./src/mem/index.js";
17
+ import { initStore } from "./src/mem/store.js";
15
18
  import { cmdPlanSeed } from "./src/plan-seed.js";
16
19
  import { cmdPriceScan } from "./src/price/index.js";
17
20
  import { cmdReport, cmdReportWeb } from "./src/report/index.js";
@@ -42,6 +45,34 @@ if (cmd === "analyze") {
42
45
  await cmdTelemetry(a);
43
46
  } else if (cmd === "init-mem") {
44
47
  cmdInitMem(a);
48
+ } else if (cmd === "mem") {
49
+ // `--mem-dir <path>` is global to `mem` and must reach both the writer
50
+ // (initStore) and the resolver behind `mem where` — parse it once and thread
51
+ // it through, never strip it and forget.
52
+ const memDirIdx = a.indexOf("--mem-dir");
53
+ let overrideMemDir: string | undefined;
54
+ let rest = a;
55
+ if (memDirIdx !== -1) {
56
+ overrideMemDir = a[memDirIdx + 1];
57
+ if (!overrideMemDir || overrideMemDir.startsWith("--")) {
58
+ console.error("fapony mem: --mem-dir needs a value");
59
+ process.exit(1);
60
+ }
61
+ if (!existsSync(overrideMemDir)) {
62
+ console.error(
63
+ `fapony mem: --mem-dir path does not exist: ${overrideMemDir}`,
64
+ );
65
+ process.exit(1);
66
+ }
67
+ rest = a.filter((_, i) => i !== memDirIdx && i !== memDirIdx + 1);
68
+ }
69
+ initStore(process.cwd(), overrideMemDir);
70
+ try {
71
+ await cmdMem(rest, overrideMemDir);
72
+ } catch (e) {
73
+ console.error(`fapony mem: ${e instanceof Error ? e.message : String(e)}`);
74
+ process.exit(1);
75
+ }
45
76
  } else if (cmd === "init") {
46
77
  await cmdInit(a);
47
78
  } else if (cmd === "install") {
@@ -75,7 +106,7 @@ if (cmd === "analyze") {
75
106
  } else {
76
107
  console.error(`fapony: unknown command "${cmd ?? ""}"`);
77
108
  console.error(
78
- "usage: fapony <setup|update|stats|telemetry|init|init-mem|install|report|report-web|usage-scan|usage-web|price-scan|analyze|debt|lint-baseline|plan-seed|review-seed|digest|mcp|hook-stop|hook-read-hint|test> [args]",
109
+ "usage: fapony <setup|update|stats|telemetry|init|init-mem|mem|install|report|report-web|usage-scan|usage-web|price-scan|analyze|debt|lint-baseline|plan-seed|review-seed|digest|mcp|hook-stop|hook-read-hint|test> [args]",
79
110
  );
80
111
  process.exit(1);
81
112
  }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "fapony",
3
- "version": "0.1.3",
4
- "description": "Measurement layer for coding agents — measure what agents do, verify what they claim. 6 MCP tools, any agent, no loop required",
3
+ "version": "0.2.0",
4
+ "description": "Measurement layer for coding agents \u2014 measure what agents do, verify what they claim. 4 MCP tools, any agent, no loop required",
5
5
  "license": "MIT",
6
6
  "author": "delamind (https://github.com/kire21b)",
7
7
  "homepage": "https://github.com/kire21b/fapony#readme",
@@ -32,6 +32,7 @@
32
32
  "typecheck": "tsc --noEmit",
33
33
  "test": "bun fapony.ts test",
34
34
  "test:fast": "SKIP_SLOW=1 bun fapony.ts test",
35
+ "test:one": "bun scripts/test-one.ts",
35
36
  "check": "bun run lint && bun run typecheck && bun fapony.ts test",
36
37
  "prepublishOnly": "bash scripts/smoke-publish.sh",
37
38
  "overview": "bun fapony.ts report-web /tmp/fapony-overview.html && open /tmp/fapony-overview.html"
@@ -36,28 +36,19 @@ You are about to move a PLAN that has been shipped to the archive.
36
36
 
37
37
  A plan that is merely *waiting* (on a person, a customer, a decision) is **not** dead and does
38
38
  not move — mark it `status: blocked` + `blocked_by: <what you are waiting for>` and leave it in
39
- `plan/`, where `plan_list` will report it as blocked instead of as backlog.
39
+ `plan/` — the frontmatter is for the next person reading the folder, and the plan stays out
40
+ of `done/`, which is what `plan-sweep` and `kickoff` go by.
40
41
 
41
- 2. **Check inbound links, then `git mv`** — `.fapony/done/` sits *beside* `.fapony/plan/`, at the
42
- same depth, so every relative link *inside* the plan (`../spec/SPEC-x.md`, `../../src/...`)
43
- keeps working untouched. Nothing to normalize. What does change is how *other plans* reach
44
- this one — a sibling reference becomes a `done/` one:
42
+ 2. **Run `plan-sweep --apply`** — this does the `git mv`, rewrites markdown links inside the
43
+ file and inbound links from other plan files, warns about plain-text mentions, and logs
44
+ a decision row — all in one call:
45
45
  ```bash
46
- grep -rln 'PLAN-foo.md' .fapony/plan/ .fapony/spec/ docs/ # who points at it
47
- git mv .fapony/plan/PLAN-foo.md .fapony/done/PLAN-foo.md # same name, same depth
48
- # in .fapony/plan/*.md: (PLAN-foo.md) -> (../done/PLAN-foo.md)
46
+ fapony mem plan-sweep <PLAN-foo.md> --apply
49
47
  ```
50
- Fewer than 5 inbound files → fix them yourself · more → report the list.
51
-
52
- **The filename gets no date prefix.** The ship date is already in the header (step 1), and
53
- duplicating it into the name buys a sortable `ls` at the price of rewriting every inbound link
54
- on every ship, forever. "What shipped on which day" is a question to derive, not to store:
55
- ```bash
56
- grep -h 'shipped' .fapony/done/*.md | sort
57
- ```
58
- If git refuses ("not under version control" — `.fapony/` is gitignored in
59
- this repo), plain `mv` instead; there's nothing to commit for an untracked path, so skip
60
- step 4 in that case.
48
+ It refuses if the file lacks a shipped header or has open mem rows (next/bug/hold/decision/note).
49
+ If git refuses ("not under version control" — `.fapony/` is gitignored in this repo), plain
50
+ `mv` instead; there's nothing to commit for an untracked path, so skip step 4 in that case.
51
+ The filename gets no date prefix — the ship date is already in the header (step 1).
61
52
 
62
53
  3. **Leave the spec where it is** — `.fapony/spec/` is a reference library, not a queue. A spec
63
54
  answers "how does this work", which is asked long after the plan that ordered it shipped, and
@@ -80,15 +71,14 @@ You are about to move a PLAN that has been shipped to the archive.
80
71
  `missing_test` / `scope_mismatch` / `unsafe_command` / `spec_gap` / `incomplete`, and
81
72
  `other` (with a `note`, which it requires) only when a real finding fits none of them
82
73
  - `note`: **omit it on a clean ship.** A verdict with no note still counts toward the plan
83
- history future drafts read ("passed round 1 before"), but only notes reach the three
84
- free-text slots `project_health_context` shows — so "clean ship" evicts a note that would
85
- have taught the next session something. Write one only when this plan hit something a
74
+ history future drafts read ("passed round 1 before"), but only notes carry prose forward — so
75
+ "clean ship" evicts a note that would have taught the next session something. Write one only when this plan hit something a
86
76
  reader could not get from the diff: what the symptom looked like, where the cause actually
87
77
  was, and the rule that follows. Standalone prose — it is read months later with no access
88
78
  to this conversation.
89
79
  - `worktree`: **absolute path** to this repo/worktree (`git rev-parse --show-toplevel`) —
90
- every other fapony tool (`fapony_usage`, `fapony_stats`, `project_health_context`)
91
- scopes by absolute path too; a bare repo name won't match those queries
80
+ every other fapony tool and query scopes by
81
+ absolute path too; a bare repo name won't match them
92
82
  - `plan`: the archived plan's path (post-move, e.g. `.fapony/done/PLAN-foo.md`)
93
83
  - `files`: repo-relative paths this plan touched (`git diff --name-only <base>..HEAD`) —
94
84
  the only input to per-file risk history; without it the verdict says something happened
@@ -101,8 +91,8 @@ You are about to move a PLAN that has been shipped to the archive.
101
91
  Input: .fapony/plan/PLAN-kickoff.md, no shipped header yet
102
92
  Steps:
103
93
  1. stamp header: > ✅ **shipped 2026-09-13** (a1b2c3)
104
- 2. inbound: README.md, .fapony/plan/PLAN-loop.md → (PLAN-kickoff.md) becomes (../done/PLAN-kickoff.md)
105
- git mv .fapony/plan/PLAN-kickoff.md .fapony/done/PLAN-kickoff.md
94
+ 2. fapony mem plan-sweep .fapony/plan/PLAN-kickoff.md --apply
95
+ → moved, links rewritten, decision logged
106
96
  3. spec: untouched, stays in .fapony/spec/
107
97
  4. commit
108
98
  5. verdict_submit(verdict="pass", reason_code="none", regime="code", worktree="/Users/you/Project/fapony/wt-fapony", plan=".fapony/done/PLAN-kickoff.md", files=["src/kickoff.ts"])
@@ -122,5 +112,5 @@ A ship worth a note looks like this instead:
122
112
 
123
113
  - No git repo / no commits (can't derive a shipped hash) → tell user: "Add header > ✅ **shipped** (<hash>) first"
124
114
  - Stamped the header yourself → always say which hash you used
125
- - Link normalize fails → report which paths normalized wrong
126
- - Too many inbound links → report full list, don't fix yourself
115
+ - plan-sweep refuses (open mem rows) → close them or use `MEM_FORCE=1`
116
+ - Too many inbound links → plan-sweep reports them; too many to fix → report the list
@@ -106,6 +106,11 @@ So the seed buys you structure; the draft budget goes on judgment:
106
106
  - **Run the Phase −1 commands for facts** when the idea needs them, and put the numbers in the
107
107
  section they answer — a number you measured beats a number the seed guessed at.
108
108
  - **Signatures live in the SPEC chunks only.** Never paste them into plan §7 — link to the spec.
109
+ - **Does this zone already owe a convention?** If the feature touches a directory, run
110
+ `fapony debt` once and read only the entries whose files overlap it. An open migration
111
+ ("36 files still throw raw errors") is a constraint for §4, not a side quest — a plan that
112
+ adds the 37th is how the debt got there. Nothing overlaps, or no `conventions.json`? Say
113
+ nothing and move on.
109
114
  - If the CLI is missing, skip silently and draft from scratch (Phase 2 as written) — never block
110
115
  on a missing tool.
111
116
 
@@ -140,8 +145,8 @@ normal — writing to the default there scatters plans into a directory nobody r
140
145
  by hand, this check is yours.)
141
146
 
142
147
  **Editing a plan someone is executing right now is a different job from drafting one.** Ask the
143
- dev, or call `plan_list` — it joins plan files against run history, so a plan with an open run is
144
- one an agent is working from this minute. When that is the case:
148
+ dev, or run `fapony mem kickoff` — it reads the same plan files and names the first unchecked
149
+ chunk, so a plan already in flight is the one you are about to edit under someone. When that is the case:
145
150
 
146
151
  - **Anything you add is an instruction, not a note.** A measured fact parked under "don't do"
147
152
  still reads as a to-do to an agent mid-execution — the numbers are what make it tempting.
@@ -180,7 +185,7 @@ only place that ordering stays true.
180
185
 
181
186
  **The TL;DR is 15 lines, hard cap, and is the only part that changes while the work is in flight**
182
187
  (tick a box, stamp a short sha). Everything below it is the agreement. A TL;DR allowed to grow
183
- becomes a second copy of the plan, and then neither copy can be trusted. `plan_list` tallies the
188
+ becomes a second copy of the plan, and then neither copy can be trusted. `fapony mem kickoff` reads the
184
189
  checkboxes in the **first `##` section only**, so section 6 stays detail rather than status.
185
190
 
186
191
  Section 6 — every step must be verifiable. Section 8 — must link back to anything it came from.
package/src/analyze.ts CHANGED
@@ -132,7 +132,19 @@ export function isTestedThroughBarrels(
132
132
  export const SCAN_EXTS = new Set([".ts", ".tsx", ".js", ".jsx"]);
133
133
 
134
134
  // Always skipped, hardcoded — no config (per plan: no .faponyignore in v1).
135
- const SKIP_DIRS = new Set(["node_modules", "dist", "build", ".git"]);
135
+ // "templates" for the same reason knip.json ignores templates/**: those files
136
+ // ship as a template copied into other repos by `fapony init-mem` and never
137
+ // have real importers here — scanning them produces false wrapper/orphan
138
+ // signals (measured: conventions-seed flagged 8 "wrappers" that were all
139
+ // src/mem/commands/*.ts helpers matched against unrelated identically-
140
+ // named calls elsewhere in the repo, e.g. "cmdNow() instead of now(").
141
+ const SKIP_DIRS = new Set([
142
+ "node_modules",
143
+ "dist",
144
+ "build",
145
+ ".git",
146
+ "templates",
147
+ ]);
136
148
 
137
149
  // A nested checkout (clone or `git worktree add`) is a different project that
138
150
  // happens to live inside this one — walking it doubles the graph and makes every
@@ -190,7 +190,6 @@ async function eslintRows(
190
190
  const rows: SeedRow[] = [];
191
191
  const configs: string[] = [];
192
192
  const skipped: string[] = [];
193
- const seen = new Set<string>();
194
193
  for (const abs of findConfigFiles(root)) {
195
194
  let raw: string;
196
195
  try {
@@ -11,7 +11,7 @@ export const DEFAULT_SPEC_DIR = ".fapony/spec";
11
11
  // Archive sits beside plan/, not inside it, so archiving never changes a file's
12
12
  // depth and its relative links survive the move untouched.
13
13
  export const DEFAULT_DONE_DIR = ".fapony/done";
14
- export const DEFAULT_MEMORY_ENTRY = ".fapony/.memory/mem.ts";
14
+ export const DEFAULT_MEM_DIR = ".fapony/.memory";
15
15
  export const DEFAULT_EVIDENCE_FILE = ".fapony/evidence.json";
16
16
 
17
17
  export const DEFAULT_CONFIG: Config = {
package/src/db/getters.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  DEFAULT_DONE_DIR,
3
3
  DEFAULT_EVIDENCE_FILE,
4
- DEFAULT_MEMORY_ENTRY,
4
+ DEFAULT_MEM_DIR,
5
5
  DEFAULT_PLAN_DIR,
6
6
  DEFAULT_SAFETY_DENY,
7
7
  DEFAULT_SPEC_DIR,
@@ -24,8 +24,8 @@ export function doneDir(config?: Config): string {
24
24
  return config?.paths?.doneDir ?? DEFAULT_DONE_DIR;
25
25
  }
26
26
 
27
- export function memoryEntry(config?: Config): string {
28
- return config?.paths?.memoryEntry ?? DEFAULT_MEMORY_ENTRY;
27
+ export function memoryDir(config?: Config): string {
28
+ return config?.paths?.memDir ?? DEFAULT_MEM_DIR;
29
29
  }
30
30
 
31
31
  export function evidenceFile(config?: Config): string {
package/src/db/store.ts CHANGED
@@ -135,48 +135,10 @@ export function addEvent(
135
135
  return Number(result.lastInsertRowid);
136
136
  }
137
137
 
138
- /**
139
- * Merge `patch` into an existing event's JSON data (keeps keys already set).
140
- * Used to complete a spawn row with bytes_out/usd after the agent finishes —
141
- * one row per spawn, timing (ts) stays at spawn start. No-op when the row
142
- * is missing or its data isn't a JSON object.
143
- */
144
- export function updateEventData(
145
- db: Database,
146
- eventId: number,
147
- patch: Record<string, unknown>,
148
- ): void {
149
- const row = db
150
- .prepare("SELECT data FROM events WHERE id = ?")
151
- .get(eventId) as { data: string | null } | null;
152
- if (!row) return;
153
- let base: Record<string, unknown> = {};
154
- try {
155
- const parsed = JSON.parse(row.data ?? "null") as unknown;
156
- if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
157
- base = parsed as Record<string, unknown>;
158
- }
159
- } catch {
160
- return;
161
- }
162
- db.prepare("UPDATE events SET data = ? WHERE id = ?").run(
163
- JSON.stringify({ ...base, ...patch }),
164
- eventId,
165
- );
166
- }
167
-
168
138
  export function getRun(db: Database, runId: number): Run | null {
169
139
  return db.prepare("SELECT * FROM runs WHERE id = ?").get(runId) as Run | null;
170
140
  }
171
141
 
172
- export function getActiveRuns(db: Database): Run[] {
173
- return db
174
- .prepare(
175
- "SELECT * FROM runs WHERE status NOT IN ('passed', 'stopped') ORDER BY id",
176
- )
177
- .all() as Run[];
178
- }
179
-
180
142
  /**
181
143
  * Latest still-open run for a worktree+plan pair, or null.
182
144
  * Used by MCP verdict_submit to bind a round-2+ verdict to the original run
@@ -221,43 +183,6 @@ export function getEvents(db: Database, runId: number): Event[] {
221
183
  .all(runId) as Event[];
222
184
  }
223
185
 
224
- /**
225
- * Pulls the most recent unresolved "gate fail" note for a worktree+mem_id pair —
226
- * i.e. the review feedback the next `fapony run` should hand back to the executor.
227
- * Only looks at the latest run for that pair; if it already passed, returns null
228
- * (nothing to carry forward).
229
- */
230
- export function getPendingFeedback(
231
- db: Database,
232
- worktree: string,
233
- memId: string,
234
- excludeRunId?: number,
235
- ): string | null {
236
- const run = db
237
- .prepare(
238
- `SELECT * FROM runs WHERE worktree = ? AND mem_id = ? AND id != ? ORDER BY id DESC LIMIT 1`,
239
- )
240
- .get(worktree, memId, excludeRunId ?? -1) as Run | null;
241
- if (run?.status !== "fixing") return null;
242
-
243
- const event = db
244
- .prepare(
245
- `SELECT * FROM events WHERE run_id = ? AND kind = 'gate' ORDER BY id DESC LIMIT 1`,
246
- )
247
- .get(run.id) as Event | null;
248
- if (!event?.data) return null;
249
-
250
- try {
251
- const parsed = JSON.parse(event.data) as {
252
- verdict?: string;
253
- note?: string;
254
- };
255
- return parsed.verdict === "fail" && parsed.note ? parsed.note : null;
256
- } catch {
257
- return null;
258
- }
259
- }
260
-
261
186
  /**
262
187
  * Merge extra fields into the most recent gate event's data JSON.
263
188
  * Used by MCP verdict_submit to add reason_code / source after gateOnce
package/src/db/types.ts CHANGED
@@ -59,7 +59,7 @@ export interface Config {
59
59
  planDir?: string;
60
60
  specDir?: string;
61
61
  doneDir?: string;
62
- memoryEntry?: string;
62
+ memDir?: string;
63
63
  evidenceFile?: string;
64
64
  } | null;
65
65
  safety?: {