claude-code-session-manager 0.40.3 → 0.41.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/dist/assets/{TiptapBody-DweEtHks.js → TiptapBody-Cr-O75Yg.js} +1 -1
  2. package/dist/assets/{index-BxVBtmjA.css → index-CKY3mHgV.css} +1 -1
  3. package/dist/assets/{index-QRU8uTeq.js → index-CV9TNjxX.js} +631 -624
  4. package/dist/index.html +2 -2
  5. package/package.json +3 -1
  6. package/plugins/session-manager-dev/skills/develop/SKILL.md +5 -5
  7. package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
  8. package/plugins/session-manager-dev/skills/explain-to-me/SKILL.md +2 -2
  9. package/plugins/session-manager-dev/skills/find-opportunity/SKILL.md +1 -1
  10. package/plugins/session-manager-dev/skills/project-status/SKILL.md +20 -34
  11. package/plugins/session-manager-dev/skills/propose-epic/SKILL.md +68 -0
  12. package/scripts/lib/watchdogHelpers.cjs +5 -5
  13. package/scripts/mint-epic.cjs +28 -0
  14. package/scripts/propose-epic.cjs +55 -0
  15. package/src/main/__tests__/epicMint.test.cjs +85 -0
  16. package/src/main/__tests__/opsErrorLog.test.cjs +87 -0
  17. package/src/main/__tests__/prdCreate.test.cjs +124 -2
  18. package/src/main/__tests__/promptSessionEvents.test.cjs +56 -2
  19. package/src/main/__tests__/promptSessionTranscript.test.cjs +0 -0
  20. package/src/main/__tests__/rcaFeedbackHook.test.cjs +43 -211
  21. package/src/main/__tests__/scheduler-archived-twin-guard.test.cjs +63 -0
  22. package/src/main/__tests__/scheduler-notify-originating-tab-transcript.test.cjs +86 -0
  23. package/src/main/__tests__/scheduler-notify-originating-tab.test.cjs +21 -0
  24. package/src/main/__tests__/scheduler-writeprd-epic-rollback.test.cjs +78 -0
  25. package/src/main/browserView.cjs +1 -1
  26. package/src/main/chatRunner.cjs +24 -47
  27. package/src/main/config.cjs +12 -6
  28. package/src/main/docEdit.cjs +3 -4
  29. package/src/main/index.cjs +12 -0
  30. package/src/main/ipcSchemas.cjs +59 -11
  31. package/src/main/lib/__tests__/opsOwnership.test.cjs +92 -0
  32. package/src/main/lib/__tests__/projectBriefCore.test.cjs +74 -0
  33. package/src/main/lib/epicMint.cjs +133 -59
  34. package/src/main/lib/opsErrorLog.cjs +96 -0
  35. package/src/main/lib/opsOwnership.cjs +169 -0
  36. package/src/main/lib/prdCreate.cjs +39 -1
  37. package/src/main/lib/prdLocations.cjs +21 -0
  38. package/src/main/lib/projectBriefCore.cjs +70 -4
  39. package/src/main/lib/queueStore.cjs +3 -0
  40. package/src/main/lib/rcaFeedbackHook.cjs +41 -55
  41. package/src/main/logs.cjs +19 -1
  42. package/src/main/projectBrief.cjs +36 -2
  43. package/src/main/promptSessionEvents.cjs +24 -2
  44. package/src/main/promptSessionTranscript.cjs +0 -0
  45. package/src/main/pty.cjs +15 -0
  46. package/src/main/queueOps.cjs +1 -1
  47. package/src/main/scheduler/prdParser.cjs +8 -0
  48. package/src/main/scheduler.cjs +158 -62
  49. package/src/main/templates/PRD_AUTHORING.md +1 -1
  50. package/src/preload/api.d.ts +83 -4
  51. package/src/preload/index.cjs +26 -5
  52. package/plugins/session-manager-dev/skills/my-feedback/SKILL.md +0 -140
  53. package/plugins/session-manager-dev/skills/optimize-kpi/SKILL.md +0 -290
  54. package/plugins/session-manager-dev/skills/process-feedback/SKILL.md +0 -265
@@ -122,6 +122,11 @@ export interface ImportRef {
122
122
  ok: boolean;
123
123
  }
124
124
 
125
+ /** Declared owner ids for the single-writer law over a project's
126
+ * session-manager-operations/ root. Mirrors OWNERS in
127
+ * src/main/lib/opsOwnership.cjs — one writer per namespace; everyone reads. */
128
+ export type OpsWriter = 'epics' | 'scheduler' | 'project-home' | 'feedback' | 'browser';
129
+
125
130
  export interface WriteResult {
126
131
  ok: boolean;
127
132
  mtimeMs: number;
@@ -415,6 +420,11 @@ export interface ScheduleJob {
415
420
  * `sourceTabId` for referential tracing even when the tab/session no
416
421
  * longer resolves. */
417
422
  sourcePromptId?: string | null;
423
+ /** The Epic this job's PRD belongs to, derived from the PRD's directory
424
+ * (`scheduler/epics/<epicId>/prds/`) — fact on disk, unlike the
425
+ * intent-carrying `sourcePromptId`/`sourceTabId`, which can disagree.
426
+ * Authoritative source for "which Epic is this from" in every UI. */
427
+ epicId?: string | null;
418
428
  }
419
429
 
420
430
  export interface SchedulePaths {
@@ -459,6 +469,8 @@ export interface PrdListItem {
459
469
  /** PRD frontmatter `sourcePromptId` — the PromptSession (Epic) id this PRD
460
470
  * was dispatched from, if any. */
461
471
  sourcePromptId?: string | null;
472
+ /** Owning Epic id, derived from the PRD's directory. See ScheduleJob.epicId. */
473
+ epicId?: string | null;
462
474
  }
463
475
 
464
476
  export interface SupervisorConfig {
@@ -874,6 +886,8 @@ export interface ProjectBriefScopeEntry {
874
886
  export interface ProjectBrief {
875
887
  version: number;
876
888
  synthesizedAt: string;
889
+ /** ISO timestamp of the last hand-edit through `projectBrief.update`, or null. */
890
+ editedAt: string | null;
877
891
  model: string;
878
892
  purpose: string;
879
893
  what: string[];
@@ -884,6 +898,15 @@ export interface ProjectBrief {
884
898
  pinned: { what: string[] | null; conventions: string[] | null };
885
899
  }
886
900
 
901
+ export interface PromptSessionTranscriptTurn {
902
+ v: 1;
903
+ epicId: string;
904
+ eventId: string | null;
905
+ role: 'user' | 'assistant';
906
+ at: string;
907
+ text: string;
908
+ }
909
+
887
910
  export interface ProjectBriefSource {
888
911
  label: string;
889
912
  detail: string;
@@ -904,6 +927,19 @@ export type ProjectBriefSetPinResult =
904
927
  | { ok: true; brief: ProjectBrief }
905
928
  | { ok: false; error: string };
906
929
 
930
+ /** Hand-edit patch over brief.json. Editing `what`/`conventions` auto-pins them. */
931
+ export interface ProjectBriefPatch {
932
+ purpose?: string;
933
+ what?: string[];
934
+ conventions?: string[];
935
+ areas?: ProjectBriefArea[];
936
+ scope?: ProjectBriefScopeEntry[];
937
+ }
938
+
939
+ export type ProjectBriefUpdateResult =
940
+ | { ok: true; brief: ProjectBrief }
941
+ | { ok: false; error: string };
942
+
907
943
  // ────────────────────────────────────────────── Per-subagent memory
908
944
  // Stored at ~/.claude/session-manager/agent-memory/<agentId>.json. Keyed by
909
945
  // agent name (the .md filename in ~/.claude/agents/), not by workspace cwd.
@@ -1184,7 +1220,7 @@ export interface SessionManagerAPI {
1184
1220
  | { ok: true; url: string; title: string; dataUrl: string }
1185
1221
  | { ok: false; error: string }
1186
1222
  >;
1187
- saveBinary: (path: string, base64: string) => Promise<{ ok: boolean; error?: string }>;
1223
+ saveBinary: (path: string, base64: string, writer?: OpsWriter) => Promise<{ ok: boolean; error?: string }>;
1188
1224
  /** Opens a native "Save As" dialog and writes `text` to the chosen path directly
1189
1225
  * (bypasses config.cjs's write-boundary check — the path is user-chosen via OS dialog). */
1190
1226
  saveRecording: (payload: { defaultName: string; text: string }) => Promise<
@@ -1247,14 +1283,22 @@ export interface SessionManagerAPI {
1247
1283
  status: () => Promise<McpStatusResult>;
1248
1284
  };
1249
1285
  logs: {
1250
- write: (scope: string, level: 'debug' | 'info' | 'warn' | 'error', message: string, meta?: unknown) => void;
1286
+ write: (
1287
+ scope: string,
1288
+ level: 'debug' | 'info' | 'warn' | 'error',
1289
+ message: string,
1290
+ meta?: unknown,
1291
+ ctx?: { cwd?: string; tabId?: string; epicId?: string; tags?: string[] },
1292
+ ) => void;
1251
1293
  dir: () => Promise<string>;
1252
1294
  };
1253
1295
  config: {
1254
1296
  readJson: (path: string) => Promise<ReadJsonResult>;
1255
1297
  readText: (path: string) => Promise<ReadTextResult>;
1256
- writeJson: (path: string, data: unknown) => Promise<WriteResult>;
1257
- writeText: (path: string, text: string) => Promise<WriteResult>;
1298
+ /** `writer` declares the owning surface for the single-writer law — required
1299
+ * when `path` is inside a project's session-manager-operations/ root. */
1300
+ writeJson: (path: string, data: unknown, writer?: OpsWriter) => Promise<WriteResult>;
1301
+ writeText: (path: string, text: string, writer?: OpsWriter) => Promise<WriteResult>;
1258
1302
  listDir: (path: string, opts?: { filesOnly?: boolean; dirsOnly?: boolean; includeHidden?: boolean }) => Promise<ListDirResult>;
1259
1303
  exists: (path: string) => Promise<boolean>;
1260
1304
  watch: (paths: string[]) => void;
@@ -1439,6 +1483,18 @@ export interface SessionManagerAPI {
1439
1483
  refresh: (cwd: string) => Promise<ProjectBriefRefreshResult>;
1440
1484
  /** Pin/unpin a synthesized block ('what' | 'conventions'), freezing or clearing its frozen copy. */
1441
1485
  setPin: (cwd: string, block: ProjectBriefPinnableBlock, pinned: boolean) => Promise<ProjectBriefSetPinResult>;
1486
+ /** Hand-edit brief.json in place — no LLM cost. Edited pinnable blocks are auto-pinned so the next refresh preserves them. */
1487
+ update: (cwd: string, patch: ProjectBriefPatch) => Promise<ProjectBriefUpdateResult>;
1488
+ };
1489
+ promptSessionTranscript: {
1490
+ /** Append one full-text turn to an Epic's durable JSONL transcript. Best-effort — resolves `{ok:false}` rather than throwing on failure. */
1491
+ append: (
1492
+ cwd: string,
1493
+ epicId: string,
1494
+ turn: { role: 'user' | 'assistant'; text: string; at?: string; eventId?: string },
1495
+ ) => Promise<{ ok: boolean }>;
1496
+ /** Read back an Epic's full-text turns (optionally capped to the last `limit`). Skips corrupt lines rather than throwing. */
1497
+ read: (cwd: string, epicId: string, limit?: number) => Promise<{ turns: PromptSessionTranscriptTurn[] }>;
1442
1498
  };
1443
1499
  agentMemory: {
1444
1500
  /** List all memory entries for one subagent. Sorted newest first. */
@@ -1522,6 +1578,29 @@ export interface SessionManagerAPI {
1522
1578
  offset?: number;
1523
1579
  }) => Promise<Exchange[]>;
1524
1580
  };
1581
+ promptSessions: {
1582
+ /** Fires when the main process appends an event to a PromptSession's
1583
+ * chain on disk (currently only the scheduler's response-event append
1584
+ * from notifyOriginatingTab) — lets an already-hydrated Epic pick it up
1585
+ * live instead of waiting for a restart. */
1586
+ onEventAppended: (handler: (e: PromptSessionEventAppendedPayload) => void) => () => void;
1587
+ };
1588
+ }
1589
+
1590
+ // ─────────────────────────────────── PromptSession event-appended broadcast
1591
+
1592
+ export interface PromptSessionEventAppendedPayload {
1593
+ cwd: string;
1594
+ promptSessionId: string;
1595
+ event: {
1596
+ id: string;
1597
+ promptSessionId: string;
1598
+ kind: 'prompt' | 'prd_created' | 'response' | 'closed';
1599
+ causedByEventId: string | null;
1600
+ at: string;
1601
+ prdSlug?: string;
1602
+ text?: string;
1603
+ };
1525
1604
  }
1526
1605
 
1527
1606
  // ────────────────────────────────────────────── Exchanges (PRD 324 read path)
@@ -82,7 +82,7 @@ contextBridge.exposeInMainWorld('api', {
82
82
  },
83
83
  captureDom: (payload) => ipcRenderer.invoke('browser:capture-dom', payload),
84
84
  captureShot: (viewId) => ipcRenderer.invoke('browser:capture-shot', { viewId }),
85
- saveBinary: (path, base64) => ipcRenderer.invoke('browser:save-binary', { path, base64 }),
85
+ saveBinary: (path, base64, writer) => ipcRenderer.invoke('browser:save-binary', { path, base64, writer }),
86
86
  saveRecording: (payload) => ipcRenderer.invoke('browser:save-recording', payload),
87
87
  replay: (payload) => ipcRenderer.invoke('browser:replay', payload),
88
88
  onReplayStep: (viewId, handler) => {
@@ -146,15 +146,18 @@ contextBridge.exposeInMainWorld('api', {
146
146
  status: () => ipcRenderer.invoke('mcp:status'),
147
147
  },
148
148
  logs: {
149
- write: (scope, level, message, meta) =>
150
- ipcRenderer.send('log:write', { scope, level, message, meta }),
149
+ // ctx is optional: { cwd, tabId, epicId, tags } — when present and
150
+ // level is 'error', main also appends a tagged line to that project's
151
+ // own session-manager-operations/logs/ (see opsErrorLog.cjs).
152
+ write: (scope, level, message, meta, ctx) =>
153
+ ipcRenderer.send('log:write', { scope, level, message, meta, ...ctx }),
151
154
  dir: () => ipcRenderer.invoke('log:dir'),
152
155
  },
153
156
  config: {
154
157
  readJson: (path) => ipcRenderer.invoke('config:read-json', { path }),
155
158
  readText: (path) => ipcRenderer.invoke('config:read-text', { path }),
156
- writeJson: (path, data) => ipcRenderer.invoke('config:write-json', { path, data }),
157
- writeText: (path, text) => ipcRenderer.invoke('config:write-text', { path, text }),
159
+ writeJson: (path, data, writer) => ipcRenderer.invoke('config:write-json', { path, data, writer }),
160
+ writeText: (path, text, writer) => ipcRenderer.invoke('config:write-text', { path, text, writer }),
158
161
  listDir: (path, opts) => ipcRenderer.invoke('config:list-dir', { path, opts }),
159
162
  exists: (path) => ipcRenderer.invoke('config:exists', { path }),
160
163
  watch: (paths) => ipcRenderer.send('config:watch', { paths }),
@@ -320,6 +323,13 @@ contextBridge.exposeInMainWorld('api', {
320
323
  get: (cwd) => ipcRenderer.invoke('project-brief:get', { cwd }),
321
324
  refresh: (cwd) => ipcRenderer.invoke('project-brief:refresh', { cwd }),
322
325
  setPin: (cwd, block, pinned) => ipcRenderer.invoke('project-brief:set-pin', { cwd, block, pinned }),
326
+ update: (cwd, patch) => ipcRenderer.invoke('project-brief:update', { cwd, patch }),
327
+ },
328
+ promptSessionTranscript: {
329
+ append: (cwd, epicId, turn) =>
330
+ ipcRenderer.invoke('promptSessionTranscript:append', { cwd, epicId, ...turn }),
331
+ read: (cwd, epicId, limit) =>
332
+ ipcRenderer.invoke('promptSessionTranscript:read', { cwd, epicId, ...(limit ? { limit } : {}) }),
323
333
  },
324
334
  agentMemory: {
325
335
  list: (agentId) => ipcRenderer.invoke('agent-memory:list', { agentId }),
@@ -440,4 +450,15 @@ contextBridge.exposeInMainWorld('api', {
440
450
  * `sessionId` filters to one session; `limit`/`offset` for pagination. */
441
451
  list: (payload) => ipcRenderer.invoke('exchanges:list', payload),
442
452
  },
453
+ promptSessions: {
454
+ /** Fires when the main process (currently only the scheduler's
455
+ * response-event append) appends an event to a PromptSession's chain on
456
+ * disk — lets an already-hydrated Epic pick it up live instead of
457
+ * waiting for a restart. */
458
+ onEventAppended: (handler) => {
459
+ const listener = (_e, payload) => handler(payload);
460
+ ipcRenderer.on('promptSession:event-appended', listener);
461
+ return () => ipcRenderer.removeListener('promptSession:event-appended', listener);
462
+ },
463
+ },
443
464
  });
@@ -1,140 +0,0 @@
1
- ---
2
- name: my-feedback
3
- description: >-
4
- File a feedback or enhancement request FROM the current project INTO another
5
- project's inbound feedback folder, following that project's own README
6
- convention. The cross-project complement of /process-feedback (which works the
7
- *current* project's inbox). Use whenever the user says "/my-feedback to X",
8
- "send feedback to X", "file a feedback item to X", "request an enhancement
9
- from X", "drop a note in X's feedback folder", or "ask X to add/fix Y". The
10
- target project MUST be named — error if it is missing. Keywords: feedback,
11
- cross-project request, enhancement request, upstream, downstream, service
12
- boundary, file feedback, request a feature, inter-service.
13
- ---
14
-
15
- # my-feedback
16
-
17
- **Role:** the *outbound* end of the agent↔agent channel — it files a request into another
18
- project's intake, where their `/process-feedback` picks it up and (if codeable) runs it
19
- through their `/develop` pipeline. It is the send side; `/process-feedback` is the receive
20
- side. **Never** cross the service boundary to fix the other project's code yourself — the
21
- deliverable is an auditable request, not a patch.
22
-
23
- Write one actionable feedback file into **another** project's intake folder, so
24
- the team that owns the boundary can act on it via their `/process-feedback`. The
25
- current project is the **From**; the named project is the **To**.
26
-
27
- This is the send side of the same channel `/process-feedback` reads. Respect the
28
- service boundary: you are *requesting* a change in their code, never reaching
29
- across to make it yourself.
30
-
31
- ## 0. Resolve the target — error if missing
32
-
33
- The invocation is `/my-feedback to <<project>>` (or "send feedback to
34
- <<project>>"). **`<<project>>` is required.**
35
-
36
- - **If no target project is named, STOP and error.** Do not guess, do not file
37
- into the current project. Print: *"Name the target project: `/my-feedback to
38
- <project>`."* Then list candidate sibling projects that have an intake folder:
39
- ```bash
40
- for d in ~/Projects/*/; do
41
- [ -d "$d/session-manager-operations/feedback" ] && echo " - $(basename "$d") (session-manager-operations/feedback/)"
42
- done
43
- ```
44
- and stop.
45
- - Resolve the path: `~/Projects/<project>/session-manager-operations/feedback/`.
46
- This is the **one** canonical folder name — do not accept or invent variants
47
- (`external-feedback/`, `feedback-inbox/`, a root-level `feedback` folder, etc.) even
48
- if a near-miss directory exists; a stray differently-named folder is drift,
49
- not a valid convention (burrow's `external-feedback/` existed for ~2.5 weeks
50
- from exactly this mistake before being merged back into `feedback` on
51
- 2026-07-10 — don't recreate it, in burrow or anywhere else). Fuzzy-match a
52
- near miss only to confirm it's actually named
53
- `session-manager-operations/feedback/`, not to accept a differently-named
54
- folder as equivalent. All session-manager per-project operations live under
55
- `session-manager-operations/`.
56
- - **If the target has no `session-manager-operations/feedback/` folder, STOP.**
57
- Don't invent one under any name — say the project doesn't accept feedback
58
- this way and ask how to proceed (it may take requests via issues, a
59
- different folder, or not at all).
60
-
61
- ## 1. Read the target's README FIRST — it is the authority
62
-
63
- **Every project's feedback folder has its own `README.md`, and the convention is
64
- unique per project** (file-naming scheme, required sections, the status-log
65
- table, the closing ritual). Read it before writing anything:
66
-
67
- ```bash
68
- cat ~/Projects/<project>/session-manager-operations/feedback/README.md
69
- ls ~/Projects/<project>/session-manager-operations/feedback/ # open items + numbering
70
- ls ~/Projects/<project>/session-manager-operations/feedback/processed/ 2>/dev/null # closed examples to match
71
- ```
72
-
73
- Read one **processed** example end-to-end to copy the house format exactly — the
74
- processed files are the calibrated bar. If the README and an example disagree,
75
- the example wins (it's what actually got accepted).
76
-
77
- ## 2. Name the file by their convention
78
-
79
- Most intakes use `YYYY-MM-DD-NN-short-slug.md` (today's date, next free `NN` for
80
- the day — check existing files so you don't collide). Use the **target's** scheme
81
- if it differs. The "From" is the **current project** — derive it from the cwd
82
- basename, not an assumption.
83
-
84
- ## 3. Write to the bar — make it closeable without a back-and-forth
85
-
86
- These hold across every intake (and most READMEs say the same):
87
-
88
- - **Cite real `file:line` from the target's current `src/`** — verify by `grep`,
89
- not memory. "You spawn a process per call — `_burrow_mcp.py:75`" beats "your
90
- transport seems slow." Quote the offending code, then name the contract/PRD it
91
- violates so the gap is self-evident.
92
- - **Lead with the cost already paid.** A concrete incident ("the silent 11-day
93
- mentions hole", "shorted into earnings on a stale zero") earns priority over a
94
- hypothetical. Tie it to money/correctness, not taste.
95
- - **Give a copy-pasteable Fix _and_ the deps to unblock it** (env var, new tool,
96
- rate bucket). Don't leave the closer to discover side-asks.
97
- - **Separate "must change" from "nice."** Credit what's already right so the file
98
- reads as calibrated, not a pile-on.
99
- - **Name the symptom and hypothesize the cause, but don't confidently assign the
100
- fix's home.** A wrong attribution stalls; "frozen at 06-08 (evidence) — maybe
101
- scoring, maybe upstream gather" closes faster.
102
- - **Flag third-service dependencies up front.** If an ask needs *another*
103
- project's tool/contract to land, say so — the closer forwards it on day one
104
- instead of discovering it mid-fix. (You can also file the dependent half
105
- directly into that third project with another `/my-feedback` pass.)
106
- - **End with concrete Asks + the target's closing ritual** (the README usually
107
- specifies marking ✅ and appending `## RESOLUTION`). State what "done" looks
108
- like, with an acceptance test per ask.
109
- - **One file = one coherent thread, scoped to the boundary.** Don't smuggle in
110
- unrelated requests.
111
-
112
- Anti-patterns that stall: vague severity ("seems slow"), no reproduction, fixes
113
- that assume internals the other team can't see, asks with no acceptance test.
114
-
115
- ## 4. Register and hand off
116
-
117
- - If the README keeps a **status-log table**, add a row for the new item (open/🆕)
118
- using their exact column format. This is usually required — the log is the
119
- durable index.
120
- - **Don't commit or push into the target repo unless the user asks.** The file in
121
- their tree is the deliverable; their team picks it up via `/process-feedback`.
122
- If you do commit, commit in the **target** repo only, and never touch their
123
- source — just the feedback file + README log row.
124
- - Report back: the file path written, the From→To, priority, and the one-line
125
- ask — so the user can relay or follow up.
126
-
127
- ## Tips
128
-
129
- - **You're the sender, not the fixer.** Resist editing the target's code to
130
- "just fix it" — that violates the boundary the channel exists to protect. The
131
- whole point is an auditable request the owner acts on.
132
- - **Mirror their tone.** A terse intake wants terse; a structured one wants every
133
- section. Match the processed examples.
134
- - **Reuse evidence you already have.** If this session surfaced the incident
135
- (logs, audit rows, live tool output), quote it verbatim — freshly-gathered
136
- `file:line` and timestamps are exactly what makes a file closeable.
137
- - **Stale-claim guard.** Feedback captures the world when written; note the
138
- as-of timestamp on live data you cite, so the closer knows what to re-verify.
139
- - **If the right home is a third project,** file there too rather than overloading
140
- one team with an ask they can only forward.
@@ -1,290 +0,0 @@
1
- ---
2
- name: optimize-kpi
3
- description: >-
4
- Project-agnostic North-Star-KPI optimization cycle. Reads the current project's
5
- declared North-Star KPI from its CLAUDE.md, measures it, audits real
6
- consumption/usage telemetry + operational logs, grades the prior iteration's
7
- filed lever, then fans out 3 INDEPENDENT recommender agents (Fable 5, max
8
- thinking) that each diagnose the gap and propose the single
9
- highest-leverage improvement — every recommendation grounded in usage + logs,
10
- not the scorecard alone; one consolidator agent merges them into ONE feedback
11
- item filed into this project's own feedback inbox plus a ranked backlog.
12
- Implementation is NEVER done inline — it always goes through the scheduler via
13
- /process-feedback → /develop. Use whenever the user says "/optimize-kpi",
14
- "optimize the KPI", "improve our north-star metric", or runs the daily KPI
15
- optimization loop. Keywords: optimize, KPI, north-star, metric, coverage,
16
- recommenders, consolidate, feedback, scheduler, attribution.
17
- model: fable
18
- ---
19
-
20
- # /optimize-kpi — measure → grade last lever → 3 recommenders → consolidate → file → scheduler
21
-
22
- You are the optimization driver for **whatever this project has declared as its
23
- North-Star KPI**. You do not implement code. Each run is one closed iteration of a
24
- *converging* loop: **discover the KPI → measure + validate → grade the prior
25
- lever → fan out 3 independent recommenders → consolidate into one feedback item +
26
- backlog → hand off to the scheduler → log**.
27
-
28
- **Non-negotiable: never implement, review, commit, publish, or reboot inline.**
29
- The single consolidated recommendation is filed into this project's own feedback
30
- inbox and dispatched through **`/process-feedback` → `/develop`**, which queues
31
- PRDs onto the session-manager scheduler and tracks them to done. Bypassing the
32
- scheduler is the one thing this skill must not do.
33
-
34
- **Autonomy contract.** This skill is usually driven headless by a cron with
35
- `--dangerously-skip-permissions`, so it must run to completion without human
36
- input. **Never call `AskUserQuestion` or otherwise block on a prompt in loop
37
- mode.** If a decision genuinely needs the operator, take the reversible default,
38
- record the open question as a `> NOTE:` line in the feedback item, and continue —
39
- do not stall the loop. (Interactive, user-invoked runs may ask; detect this by
40
- whether a human is in the turn.)
41
-
42
- ## Step 0 — Discover the project's North-Star KPI (do not assume)
43
-
44
- The KPI lives in the project's mission statement: read `./CLAUDE.md` and find the
45
- section whose heading names the **North-Star KPI** (e.g. `## Objective &
46
- North-Star KPI`). That section is the authority. Extract, by reading it and the
47
- files it points to:
48
-
49
- - **The KPI statement** — what is being optimized, the target (usually 100% / a
50
- ceiling), and the precise definition doc it links (e.g. `docs/<kpi>.md`).
51
- - **The measurement command** — the scorecard/metric script it names (e.g.
52
- `scripts/coverage_scorecard.py`). Prefer a `--json` form and a `--trend N`
53
- form if they exist.
54
- - **The driving pipeline(s)** — which scheduled job(s) actually move the metric
55
- (the gatherer/worker named in the KPI section or registry). You need this for
56
- the validity gate in Step 1.
57
- - **The usage / consumption audit** — how the project's *outputs* are consumed
58
- and the operational logs that show whether the machine is healthy. The CLAUDE.md
59
- observability/telemetry pointer names them (e.g. a usage-audit script over a
60
- request log, a metrics endpoint, the run-history DB, `data/logs/`). Capture the
61
- exact audit command(s). If the project exposes **no** consumption telemetry at
62
- all, that itself is a finding — recommend adding it.
63
- - **The levers** — the files the section names as the knobs (tiers/config,
64
- selection, throughput/cadence/registry, etc.).
65
- - **The feedback intake** — `session-manager-operations/feedback/` at the repo
66
- root (the one canonical name — do not treat a differently-named folder as
67
- equivalent), and its `README.md` convention.
68
-
69
- **If CLAUDE.md declares no North-Star KPI section, STOP** and tell the user this
70
- project hasn't declared one — the KPI belongs in CLAUDE.md's mission statement
71
- (that's the convention this skill reads), and ask them to add it before running.
72
-
73
- Capture the discovered KPI statement, measurement command, driving pipeline,
74
- lever files, and intake path — you pass all of them verbatim into the recommender
75
- agents so they work from the project's own definition, not your guess.
76
-
77
- ## Step 1 — Measure + validity gate (don't tune a corpse)
78
-
79
- Run the discovered measurement command in both its snapshot and trend forms (e.g.
80
- `<scorecard> --json` and `<scorecard> --trend 14`). Read the JSON outputs. Record:
81
- today's KPI value, any floor/SLO breaches, per-segment breakdown, the
82
- worst-performing units, and the trend direction over the last several days. This
83
- evidence block is the shared input every recommender receives.
84
-
85
- **Validity gate — before diagnosing, prove the metric is real.** A low KPI has
86
- two very different causes: a *mis-set knob* (a tuning problem) or *the driving
87
- pipeline didn't run / is wedged* (an ops problem). Confirm the Step-0 driving
88
- pipeline actually produced work in the measurement window — check the
89
- orchestrator/run history (last successful run, duration, whether it hung to its
90
- max-duration cap or held a lock). If the pipeline **did not run a healthy cycle
91
- in the window**, the metric is an artifact:
92
-
93
- - Do **not** run the recommender fan-out. A selection/tiering "fix" against a
94
- dead pipeline is noise.
95
- - File a single feedback item whose Ask is **"restore the driving pipeline"**
96
- (cite the stuck/missing runs), hand it to the scheduler (Step 5), log it, and
97
- stop the iteration there.
98
-
99
- Only when the metric reflects a pipeline that genuinely ran do you proceed to the
100
- tuning loop.
101
-
102
- ## Step 1b — Audit usage + logs (MANDATORY evidence, not optional)
103
-
104
- The scorecard says *how high* the KPI is; it never says *why*. The why lives in
105
- how the project's outputs are actually consumed and what the logs are screaming.
106
- **This step is required every run — a recommendation that cites only the
107
- scorecard is rejected in Step 4.**
108
-
109
- - **Usage / consumption audit.** Run the Step-0 usage-audit command (for a
110
- contract/MCP project, the per-consumer + per-tool telemetry; for others, the
111
- request/access log or metrics endpoint). Extract: who the consumers are, the
112
- call distribution across outputs, **dark outputs** (shipped but never consumed),
113
- **error hotspots** (esp. a write/auth path failing silently), and **latency/SLO
114
- breaches**. Consumption shape tells you what's load-bearing vs dead weight — you
115
- do not optimize what nobody reads, and a silently-failing consumer is often the
116
- real KPI gap.
117
- - **Operational-log audit.** Scan the run history + `data/logs/` (or the
118
- project's log sink) for failed/stuck runs, repeated errors, max-duration
119
- overruns, and lock contention in the measurement window. These are frequently
120
- the *direct* root cause of a depressed KPI (e.g. the gatherer hitting its
121
- duration cap → fewer units visited → coverage falls) — and they're invisible to
122
- the scorecard.
123
-
124
- Produce a compact **usage+log evidence block** alongside the Step-1 scorecard
125
- block. Both are handed to every recommender. If the audit surfaces an ops failure
126
- that is *itself* the dominant KPI cause (a dead consumer, a crashing pipeline),
127
- treat it like the validity gate: the recommendation is "fix that," and you may
128
- skip the tuning fan-out.
129
-
130
- ## Step 2 — Grade the prior lever + read in-flight work (close the loop)
131
-
132
- Before proposing anything new, find out what the *last* iteration did and whether
133
- it worked. Open-loop optimizers don't converge — this step is what makes the loop
134
- learn.
135
-
136
- - **Read the optimization log** (`docs/kpi-optimization-log.md`, if present) for
137
- the last filed lever, its predicted effect, and its measure-by date.
138
- - **Check whether it shipped:** look at the feedback folder (is last run's item
139
- ✅/archived?) and the scheduler's PRD history for the ids it queued. Status:
140
- *not yet queued* / *queued, running* / *shipped on `<date>`* / *failed*.
141
- - **Grade it against the metric:** if it shipped, compare the KPI/breach numbers
142
- before vs after its ship date using the Step-1 trend. Did it move the metric in
143
- the predicted direction and magnitude? Record `helped` / `no-effect` /
144
- `regressed` / `too-early-to-tell`.
145
- - **Build the in-flight + cooldown set:** list every lever currently *queued or
146
- awaiting its first post-ship measurement*. These are **off-limits this run** —
147
- do not let the recommenders or consolidator re-file a lever that is already in
148
- flight or whose effect hasn't been measured yet (cooldown). Re-filing the same
149
- ask stacks duplicate PRDs on the scheduler and never learns.
150
-
151
- Carry forward into Step 3: `{last lever, its grade, why it under/over-performed,
152
- the in-flight/cooldown lever set}`.
153
-
154
- ## Step 3 — Fan out 3 INDEPENDENT recommenders (parallel)
155
-
156
- Spawn **3 Agent calls in a single message** so they run concurrently and blind to
157
- each other. Each is a peer doing the full diagnosis independently — divergence is
158
- the point; do not coordinate them. (Default panel = 3; you may scale to 4–5 when
159
- the gap is large and token budget allows, or drop to 2 for a tiny gap — note the
160
- choice.)
161
-
162
- - **Model:** `fable` (Fable 5) with **failover to `opus`, then `sonnet`** — launch
163
- each agent with `model: fable`; if an agent dies on a model/availability error,
164
- re-spawn that one with `model: opus`, and `sonnet` only if opus is also
165
- unavailable. A **mixed panel** (some fable, some opus/sonnet) is fine — don't
166
- block waiting for a uniform fleet. Instruct each to think at **maximum depth**
167
- before answering.
168
- - **Prompt (identical for all 3, vary only an angle hint):** give each the KPI
169
- statement, the definition doc path, the Step-1 scorecard block, the **Step-1b
170
- usage+log evidence block**, the **Step-2 prior-lever grade + in-flight/cooldown
171
- set**, the lever files, and the project's standing design directives from the
172
- KPI section. Ask each to:
173
- 1. Diagnose *why* the KPI is below target — name the single root cause with the
174
- highest expected KPI delta. **Ground the diagnosis in the usage+log evidence,
175
- not the scorecard alone:** cite the specific consumer behaviour, dark output,
176
- error hotspot, or failing/over-running run that supports the root cause. A
177
- diagnosis with no usage/log citation is not acceptable. Account for the prior
178
- lever's result: if it under-performed, say why and whether to escalate or
179
- abandon it.
180
- 2. Propose ONE concrete, highest-leverage change as numbered implementation
181
- steps an executor could follow without further design: exact file(s), the
182
- change, the lever value, the **expected KPI/breach delta as a falsifiable
183
- numeric target with a measure-by horizon** (e.g. "floor breaches 82→<40
184
- within 2 days"), and how the next measurement run will confirm it. Quality
185
- bar / service contracts stay fixed — never raise the KPI by lowering a
186
- quality gate. **Must not** be a lever in the in-flight/cooldown set.
187
- 3. State assumptions and the one risk that would make it backfire.
188
- Give each a different framing nudge so they don't converge: e.g. agent 1
189
- "selection/throughput first", agent 2 "tiering/cadence/targets-achievability
190
- first", agent 3 "consolidation/architecture first". The nudge biases the lens,
191
- not the conclusion — each still considers all levers.
192
- - **Output:** each agent **writes its recommendation to a temp file** and returns
193
- the path. Use a per-run temp dir derived from the project + date, e.g.
194
- `downloads/optimize-kpi/<YYYY-MM-DD>/agent-{1,2,3}.md` (create it; fall back to
195
- `/tmp/optimize-kpi-<date>/` if `downloads/` doesn't exist). One file per agent,
196
- full reasoning + the numbered steps. (Old per-run dirs are disposable — GC dirs
197
- older than ~14 days when you create today's.)
198
-
199
- ## Step 4 — Consolidate into ONE recommendation + a ranked backlog (single agent)
200
-
201
- Spawn **one** consolidator agent (same `fable`→`opus`→`sonnet` failover, max thinking).
202
- Give it the 3 temp files **and** the Step-2 in-flight/cooldown set. It must:
203
-
204
- - Read all three, dedupe overlapping ideas, and **score** the distinct proposals
205
- by expected KPI delta × achievability × reversibility.
206
- - **Reject any proposal not grounded in the usage+log evidence** — a root cause
207
- citing only the scorecard is not eligible. The winning lever must trace to an
208
- observed consumer behaviour, dark output, error hotspot, or failing/over-running
209
- run.
210
- - **Drop any proposal that collides with the in-flight/cooldown set** (already
211
- queued or awaiting measurement) — those are not eligible this run.
212
- - **Pick the single highest-leverage *eligible* lever** to change this iteration
213
- (resist bundling — one lever per iteration), grafting the best supporting ideas
214
- from the runners-up into the plan where they strengthen it.
215
- - Verify targets are achievable for real throughput — if a proposal sets targets
216
- the system can't meet, say so and adjust rather than let the KPI lie.
217
- - **Route the lever by ownership first.** If the winning lever's root cause
218
- belongs to an **upstream/downstream service** — the data source hasn't gathered
219
- the inputs, an upstream contract drifted, a consumer needs a change — the
220
- deliverable is a **cross-project filing via `/my-feedback <project>`** into THAT
221
- service's intake, not a this-project item this repo can't action. (Canonical
222
- case: signal-builder's sellable-KPI binding constraint is frequently the
223
- `acquisition` cohort — tickers Burrow simply hasn't gathered — which is filed to
224
- `burrow`, per this project's CLAUDE.md North-Star section.) A this-project-owned
225
- lever is filed here (next bullet); a **mixed** lever splits — the local half here
226
- via `/develop`, the upstream half via `/my-feedback`, each cross-referencing the
227
- other. Leaving an upstream-owned lever buried as a local item it can't fix is the
228
- failure mode this rule exists to prevent.
229
- - **Write ONE consolidated feedback item** (for the this-project-owned half) into
230
- this project's intake folder, named by the folder README's convention
231
- (`session-manager-operations/feedback/<YYYY-MM-DD>-NN-kpi-optimization.md`, next free `NN`), with: title,
232
- **From:** `optimize-kpi loop`, date (PT), priority + why, **TL;DR**, **Evidence**
233
- (cite the Step-1 scorecard numbers, the **Step-1b usage+log findings** the root
234
- cause rests on, *and* the Step-2 prior-lever grade), **Why it matters**
235
- (expected KPI gain), and **Ask** (the numbered implementation steps from the
236
- winning proposal). The Ask **must** carry a falsifiable acceptance test: the
237
- numeric success criterion + measure-by date the next run will grade it against.
238
- Note which of the 3 agents it drew from and why it rejected the others.
239
- - **Write/refresh a ranked backlog** at `docs/kpi-optimization-backlog.md`: the
240
- eligible-but-not-chosen levers, ranked, each with its one-line rationale and
241
- expected delta. The next iteration reads this first and can pull the next-best
242
- lever without re-deriving from scratch. Remove entries that shipped or went
243
- stale.
244
-
245
- The consolidator writes the feedback file + backlog; it does **not** touch
246
- project source.
247
-
248
- ## Step 5 — Hand off to the scheduler (do NOT implement)
249
-
250
- Dispatch the consolidated item through the scheduler pipeline:
251
-
252
- - Invoke **`/process-feedback`**, scoped to the one file you just wrote (name it
253
- explicitly — never let it sweep unrelated open items). `/process-feedback`
254
- evaluates it and queues the codeable work as PRDs via **`/develop`** onto the
255
- session-manager scheduler, then tracks those PRDs to completion.
256
- - This skill does **not** edit code, run `/code-review`/`/security-review`, bump
257
- `VERSION`, commit, or reboot anything. All of that happens as the scheduled
258
- PRDs run headlessly. Your job ends at "filed + dispatched + tracked".
259
-
260
- ## Step 6 — Log the iteration
261
-
262
- Append one line to `docs/kpi-optimization-log.md` (create if missing):
263
- `<date> | KPI <today> (trend <dir>) | breaches <n> | prior lever: <one-line> → <helped|no-effect|regressed|too-early> | lever filed: <one-line> (target: <numeric>, by <date>) | feedback: <file> | PRDs: <ids or "queued">`.
264
- This is the running record — and the input Step 2 grades next time.
265
-
266
- ## Stop conditions
267
-
268
- - **No KPI declared** (Step 0) → STOP, ask the user to add it to CLAUDE.md.
269
- - **Validity gate fails** (Step 1) → file the "restore the driving pipeline" item,
270
- hand off, log, and stop the iteration — skip the tuning fan-out entirely.
271
- - **Step-1b audit finds an ops failure that is itself the dominant KPI cause**
272
- (a dead/failing consumer, a crashing or max-duration-overrunning pipeline) →
273
- file *that* fix and skip the tuning fan-out; it outranks any tuning lever.
274
- - **KPI already at/above target** with zero breaches for 2 consecutive days →
275
- file a *consolidation / quality-hardening* recommendation instead of a coverage
276
- one (move toward the architecture end-state the KPI section describes).
277
- - **Everything eligible is in-flight/cooldown** (Step 2 leaves no eligible lever)
278
- → do not invent a duplicate. Log "no eligible lever — N in flight" and stop;
279
- next run grades them.
280
- - **Measurement, the agents, or `/process-feedback` fail** → STOP, leave the tree
281
- untouched (you never modified source anyway), and report what broke. Never file
282
- a recommendation built on a failed measurement.
283
-
284
- ## Scheduling
285
-
286
- Typically run as a daily `/loop` (the project may install a cron that drives this
287
- skill headless). Each firing runs Steps 0–6 once. Because implementation is
288
- delegated to the scheduler, this skill stays fast and side-effect-light: its only
289
- writes to the repo are the temp recommender files, the one feedback item, the
290
- backlog, and the log line.