eklavya 1.26.0 → 1.27.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +48 -42
  2. package/dist/artifacts.js +214 -0
  3. package/dist/artifacts.js.map +1 -0
  4. package/dist/assets/artifact-template.html +144 -0
  5. package/dist/assets/dashboard.html +131 -11
  6. package/dist/assets/tokens.css +6 -4
  7. package/dist/assets/tutor/SKILL.md +7 -0
  8. package/dist/cli-memory.js +6 -3
  9. package/dist/cli-memory.js.map +1 -1
  10. package/dist/cli.js +85 -2
  11. package/dist/cli.js.map +1 -1
  12. package/dist/config.js +3 -0
  13. package/dist/config.js.map +1 -1
  14. package/dist/dashboard.js +41 -1
  15. package/dist/dashboard.js.map +1 -1
  16. package/dist/hooks/memory-lib.js +1 -1
  17. package/dist/hooks/memory-lib.js.map +1 -1
  18. package/dist/hooks/session-start.js +10 -2
  19. package/dist/hooks/session-start.js.map +1 -1
  20. package/dist/install-lock.js +155 -0
  21. package/dist/install-lock.js.map +1 -0
  22. package/dist/install.js +64 -45
  23. package/dist/install.js.map +1 -1
  24. package/dist/memory/reservation.js +15 -5
  25. package/dist/memory/reservation.js.map +1 -1
  26. package/dist/memory/worker.js +72 -31
  27. package/dist/memory/worker.js.map +1 -1
  28. package/dist/migrations/015_event_link_indexes.sql +24 -0
  29. package/dist/paths.js +8 -0
  30. package/dist/paths.js.map +1 -1
  31. package/dist/plugin/.claude-plugin/plugin.json +1 -1
  32. package/dist/plugin/agents/explainer.md +57 -0
  33. package/dist/plugin/agents/tutor.md +7 -0
  34. package/dist/plugin/cli/CLAUDE.md +60 -108
  35. package/dist/plugin/hooks/CLAUDE.md +118 -357
  36. package/dist/plugin/hooks/run.mjs +104 -21
  37. package/dist/plugin/skills/CLAUDE.md +106 -291
  38. package/dist/plugin/skills/tutor/SKILL.md +7 -0
  39. package/dist/tools/config_tools.js +4 -0
  40. package/dist/tools/config_tools.js.map +1 -1
  41. package/dist/tools/record_attempt.js +25 -1
  42. package/dist/tools/record_attempt.js.map +1 -1
  43. package/dist/update.js +7 -83
  44. package/dist/update.js.map +1 -1
  45. package/dist/user-skill/eklavya/SKILL.md +8 -0
  46. package/dist/user-skill/eklavya-artifacts/SKILL.md +135 -0
  47. package/package.json +1 -1
@@ -33,12 +33,16 @@
33
33
  * The same heal keeps rule 3 current. The plugin updates itself through Claude
34
34
  * Code (a marketplace pull, `/plugin update`); the runtime only moves when npm
35
35
  * runs. So when the runtime is older than the plugin pins, this run still uses
36
- * it — never blocking — and a background install brings it up to the pin.
37
- *
36
+ * it — never blocking — and a background install brings it up to the pin,
37
+ * unless the machine opted out with `auto_update: false`. A MISSING runtime is
38
+ * installed whatever that setting says: without it nothing works, and
39
+ * installing the plugin was the request for it.
40
+ *
38
41
  * Hard rule: a hook must never break a session. Everything here
39
42
  * fails to exit 0 in silence.
40
43
  */
41
- import { existsSync, mkdirSync, readFileSync, renameSync, rmSync, statSync, writeFileSync } from 'node:fs';
44
+ import { existsSync, linkSync, mkdirSync, readFileSync, renameSync, rmSync, statSync, writeFileSync } from 'node:fs';
45
+ import { randomUUID } from 'node:crypto';
42
46
  import { homedir } from 'node:os';
43
47
  import path from 'node:path';
44
48
  import { fileURLToPath, pathToFileURL } from 'node:url';
@@ -98,11 +102,27 @@ function olderThan(a, b) {
98
102
  */
99
103
  function healIfBehind(entry) {
100
104
  if (!entry.startsWith(path.join(runtimeHome, 'node_modules', 'eklavya') + path.sep)) return;
105
+ if (!autoUpdateEnabled()) return;
101
106
  const installed = runtimeVersion();
102
107
  const pinned = pinnedVersion();
103
108
  if (installed && pinned && olderThan(installed, pinned)) healInBackground();
104
109
  }
105
110
 
111
+ /**
112
+ * `auto_update` from the machine's config file — the global one only, as
113
+ * `autoUpdateEnabled` in the runtime's `update.ts` reads it: a project cannot
114
+ * opt the machine in or out. One small JSON read; anything unreadable is the
115
+ * default, which is on.
116
+ */
117
+ function autoUpdateEnabled() {
118
+ try {
119
+ const home = process.env.EKLAVYA_HOME ?? path.join(homedir(), '.eklavya');
120
+ return JSON.parse(readFileSync(path.join(home, 'config.json'), 'utf8')).auto_update !== false;
121
+ } catch {
122
+ return true;
123
+ }
124
+ }
125
+
106
126
  /** The compiled entry point for `name`, or null if no build is reachable. */
107
127
  function resolveEntry() {
108
128
  const relative = name === 'server' ? ['dist', 'server.js'] : ['dist', 'hooks', `${name}.js`];
@@ -141,7 +161,9 @@ function healInBackground() {
141
161
  } catch {
142
162
  return;
143
163
  }
144
- if (!claimHeal(path.join(runtimeHome, '.installing'))) return;
164
+ const stamp = path.join(runtimeHome, '.installing');
165
+ const token = claimHeal(stamp);
166
+ if (!token) return;
145
167
 
146
168
  try {
147
169
  const npm = process.platform === 'win32' ? 'npm.cmd' : 'npm';
@@ -155,44 +177,105 @@ function healInBackground() {
155
177
  // server this run goes on to import.
156
178
  child.on('error', () => {});
157
179
  child.unref();
180
+ // The claim now belongs to npm, which outlives this process: it is live
181
+ // exactly as long as npm runs. Nobody breaks a claim whose pid is alive,
182
+ // so the stamp is still this run's to rewrite.
183
+ if (child.pid) writeClaim(stamp, { pid: child.pid, token, at: new Date().toISOString() });
158
184
  } catch {
159
185
  /* A failed heal is a slow install, not a broken session. */
160
186
  }
161
187
  }
162
188
 
163
- const HEAL_CLAIM_MS = 60 * 60 * 1000;
189
+ /** Shared with the runtime's `install-lock.ts`, which holds the same rules. */
190
+ const LOCK_TTL_MS = 60 * 60 * 1000;
191
+ const LEGACY_DATE_MS = 10 * 60 * 1000;
192
+ /** How often the heal may start npm, finished or not. */
193
+ const HEAL_EVERY_MS = 60 * 60 * 1000;
194
+
195
+ function ownerOf(text) {
196
+ const body = text.trim();
197
+ if (/^\d+$/.test(body)) return { pid: Number(body), token: null };
198
+ try {
199
+ const value = JSON.parse(body);
200
+ return { pid: typeof value.pid === 'number' ? value.pid : null, token: typeof value.token === 'string' ? value.token : null };
201
+ } catch {
202
+ return { pid: null, token: null };
203
+ }
204
+ }
205
+
206
+ function pidAlive(pid) {
207
+ try {
208
+ process.kill(pid, 0);
209
+ return true;
210
+ } catch (err) {
211
+ return err.code === 'EPERM';
212
+ }
213
+ }
214
+
215
+ /** Is this stamp somebody's live claim on the runtime? */
216
+ function liveClaim(file, text) {
217
+ const age = Date.now() - statSync(file).mtimeMs;
218
+ if (age >= LOCK_TTL_MS) return false;
219
+ const { pid } = ownerOf(text);
220
+ return pid !== null ? pidAlive(pid) : age < LEGACY_DATE_MS;
221
+ }
222
+
223
+ /** Replaces the stamp in one step, so a reader never sees half of it. */
224
+ function writeClaim(stamp, owner) {
225
+ try {
226
+ const tmp = `${stamp}.${process.pid}.tmp`;
227
+ writeFileSync(tmp, JSON.stringify(owner));
228
+ renameSync(tmp, stamp);
229
+ } catch {
230
+ /* It still names this process, which is gone soon: the next claim breaks it. */
231
+ }
232
+ }
164
233
 
165
234
  /**
166
- * Take the `.installing` stamp, or say someone else holds it. The session's
167
- * hooks and its server start within milliseconds of each other, so a stat
168
- * followed by a write lets several through: creating with `wx` is the one
169
- * step only one process can win. A stamp older than the hour is taken by
170
- * renaming it away — again one winner — and put back if it turned out to be
171
- * a fresh claim that landed in between.
235
+ * Take the runtime lock — the `.installing` stamp `eklavya install` and the
236
+ * updater take too — and return its token, or null. The session's hooks and
237
+ * its server start within milliseconds of each other, so a stat followed by a
238
+ * write lets several through: creating with `wx` is the one step only one
239
+ * process can win.
240
+ *
241
+ * A stamp is broken only when it is not a live claim — its pid is dead, or it
242
+ * is past the hour — and, for the heal alone, only once it is an hour old: the
243
+ * stamp a finished or failed heal leaves behind is also what keeps a broken
244
+ * npm from being retried on every hook. Breaking is a rename (one winner), and
245
+ * a fresh claim renamed by mistake is put back with `link`, which never
246
+ * overwrites.
172
247
  */
173
248
  function claimHeal(stamp) {
249
+ const token = randomUUID();
174
250
  const create = () => {
175
251
  try {
176
- writeFileSync(stamp, new Date().toISOString(), { flag: 'wx' });
177
- return true;
252
+ writeFileSync(stamp, JSON.stringify({ pid: process.pid, token, at: new Date().toISOString() }), { flag: 'wx' });
253
+ return token;
178
254
  } catch {
179
- return false;
255
+ return null;
180
256
  }
181
257
  };
182
- if (create()) return true;
258
+ if (create()) return token;
259
+ const taken = `${stamp}.${process.pid}.${token}`;
183
260
  try {
184
- if (Date.now() - statSync(stamp).mtimeMs < HEAL_CLAIM_MS) return false;
185
- const taken = `${stamp}.${process.pid}`;
261
+ const judged = readFileSync(stamp, 'utf8');
262
+ if (liveClaim(stamp, judged) || Date.now() - statSync(stamp).mtimeMs < HEAL_EVERY_MS) return null;
186
263
  renameSync(stamp, taken);
187
- if (Date.now() - statSync(taken).mtimeMs < HEAL_CLAIM_MS) {
188
- renameSync(taken, stamp);
189
- return false;
264
+ if (readFileSync(taken, 'utf8') !== judged) {
265
+ try {
266
+ linkSync(taken, stamp);
267
+ } catch {
268
+ /* someone else holds it now; theirs stands */
269
+ }
270
+ rmSync(taken, { force: true });
271
+ return null;
190
272
  }
191
273
  rmSync(taken, { force: true });
192
274
  } catch {
193
275
  // Cannot even read or move the stamp, so we cannot bound the retries.
194
276
  // Doing nothing is the safe failure: the explicit installer still works.
195
- return false;
277
+ rmSync(taken, { force: true });
278
+ return null;
196
279
  }
197
280
  return create();
198
281
  }
@@ -1,297 +1,112 @@
1
- # Editing a skill
1
+ # Editing skills and agent instructions
2
2
 
3
- `skills/` is where Eklavya's behaviour actually lives. The MCP server holds the
4
- data and the arithmetic; these files hold everything the model does with it.
5
- A wrong sentence here is a wrong product, and nothing fails loudly when it is.
3
+ Skills define model behavior; the runtime owns data and arithmetic. A stale
4
+ sentence can change the product without raising an error. Read the root
5
+ `CLAUDE.md`, then verify instructions against the tool implementation.
6
6
 
7
- ## What a skill is, mechanically
7
+ ## Entry points and ownership
8
8
 
9
- One directory, one `SKILL.md`, YAML frontmatter with `name` and `description`.
10
- **`disable-model-invocation: true` is what turns a skill into a
11
- `/eklavya:<name>` slash command** — the model can no longer load it on its own,
12
- and the developer types it. Without that line the skill is model-invocable only.
13
-
14
- Nine have it, and they are the nine slash commands:
15
-
16
- `gate`, `learn`, `level`, `memory`, `mode`, `pack`, `progress`, `quiz`, `setup`.
17
-
18
- `skills/tutor/SKILL.md` deliberately does not. It is the pedagogy — one
19
- question at a time, honest grading, never the same question twice — and every
20
- other skill defers to it rather than restating it. `quiz` says "follow the
21
- `tutor` skill for how to ask and grade"; keep it that way. Pedagogy duplicated
22
- into a command skill is pedagogy that drifts from the one place Cursor reads.
23
-
24
- ## The description is a trigger, not a summary
25
-
26
- For a model-invocable skill the `description` is the only part always in
27
- context; the body is read only once the model has decided to load it. So a
28
- description that summarises the workflow becomes a shortcut the model takes
29
- *instead of* reading the body — it answers from the summary and never opens the
30
- file. The body becomes documentation nobody reads.
31
-
32
- `skills/tutor/SKILL.md` had exactly that shape. It read "Use while implementing
33
- any non-trivial task (to log the concepts it touches), and whenever quizzing,
34
- grading, or explaining" — three steps named, and nowhere in it the words *one
35
- question, mid-task*. Which is the failure the interleaved cadence had to fix:
36
- questions arriving in a pile at the end of the work.
37
-
38
- So, for `tutor` and anything else without `disable-model-invocation`:
39
- **triggering conditions only**, third person, opening "Use when". Never the
40
- number of questions, never the order of the tool calls, never the grading.
41
- Those live in the body, which is where the model has to go to get them.
42
-
43
- The nine slash commands are exempt, and it is not a technicality:
44
- `disable-model-invocation: true` means the model never matches on their
45
- description at all. The developer types the command and the description is its
46
- one line of help, so those should say what they do. `agents/tutor.md` keeps one
47
- identity clause for the same kind of reason — a subagent is picked from a roster
48
- by what it is, not loaded by trigger.
49
-
50
- ## Three directories, and why they are not one
51
-
52
- - **`skills/`** ships inside the plugin. Slash commands plus `tutor`.
53
- - **`user-skill/eklavya/`** is copied to `~/.claude/skills/eklavya/` by
54
- `npx eklavya install` (`installSkill()` in `mcp/src/install.ts`). It is
55
- model-invocable, so plain chat — "Eklavya is quizzing me too much" — reaches
56
- it, and it works where the plugin is not loaded. **Never move it under
57
- `skills/`**: it would then register twice, once per surface. It drives the
58
- `eklavya` CLI and the read/write config tools; it does not teach or quiz.
59
- - **`agents/tutor.md`** is the subagent. It has the Eklavya MCP tools and
60
- read-only file access — and **no `AskUserQuestion`** — so it renders the four
61
- options as lettered text. Any change to
62
- `skills/tutor/references/writing-mcq.md` has to hold for a plain-text
63
- renderer too.
64
-
65
- ## A skill is a prompt, but it is also an API client
66
-
67
- Twenty tools, all in `mcp/src/tools/`. Nine for learning:
68
- `get_learner_profile`, `log_session_concepts`, `get_session_quiz_plan`,
69
- `record_attempt`, `get_gate_status`, `upsert_concepts`, `get_concept_graph`,
70
- `get_config`, `set_config`. Eleven for memory: `memory_search`, `memory_get`,
71
- `memory_timeline`, `memory_file_history`, `memory_status`, `memory_write`,
72
- `memory_correct`, `memory_delete`, `memory_collections`, `code_outline`,
73
- `code_find_symbol`.
74
-
75
- The memory tools are two-stage on purpose: the index tools return identifiers
76
- and titles, and `memory_get` is the only one that returns a narrative. A skill
77
- that tells the model to hydrate everything a search returned spends exactly
78
- the context the feature exists to save. Say "search, choose, then get".
79
-
80
- **Before you write "call `X` with `Y`", open `mcp/src/tools/<X>.ts`.** Check
81
- `Y` is in the `inputSchema`, and check the field you are telling the model to
82
- read is in what the handler returns. A skill that names a field the server
83
- never returns fails silently — the model improvises a plausible value and the
84
- developer sees a confident number nobody computed.
85
-
86
- `config_tools.ts` holds `get_config` and `set_config`, and the memory tools are
87
- grouped the same way — `memory_read_tools.ts`, `memory_write_tools.ts`,
88
- `code_tools.ts`, `collection_tools.ts`. The nine learning tools are one file
89
- per tool name.
90
-
91
- ### Not every tool takes `session_id`
9
+ Each skill has a directory and `SKILL.md` with `name` and `description`.
10
+ `disable-model-invocation: true` makes the entry user-invoked only. The shipped
11
+ slash commands are `gate`, `learn`, `level`, `memory`, `mode`, `pack`, `progress`,
12
+ `quiz` and `setup`. Recount from frontmatter when changing this inventory.
92
13
 
14
+ | Location | Responsibility |
15
+ |---|---|
16
+ | `skills/tutor/` | Shared pedagogy; commands defer to it rather than duplicate it |
17
+ | `user-skill/eklavya/` | Plain-chat configuration and CLI help, installed to `~/.claude/skills/`; does not teach |
18
+ | `user-skill/eklavya-artifacts/` | Create pages through `eklavya artifacts new`, which owns paths, metadata and template |
19
+ | `agents/tutor.md` | Read-only file access and selected learning/memory tools; teaches builder-logged concepts and uses lettered text without `AskUserQuestion` |
20
+ | `agents/explainer.md` | Background artifact writer; no Eklavya MCP tools, grading or concept logging |
21
+
22
+ Never move `user-skill/` entries into `skills/`: they would register twice.
23
+ Keep artifact design instructions and the explainer's summary aligned.
24
+
25
+ Model-invocable descriptions should state triggering conditions, in third person,
26
+ starting “Use when”; put workflow, tool order and grading in the body. A summary
27
+ in the description invites the model to skip reading the rules. User-invoked
28
+ command descriptions instead serve as concise help. Agent descriptions also
29
+ need their role so they can be selected from a roster.
30
+
31
+ ## Check the API before writing instructions
32
+
33
+ Open the relevant file in `mcp/src/tools/` before saying “call X with Y”. Verify
34
+ the input schema and every returned field the instruction uses. Learning tools
35
+ mostly have individual files; config, memory read/write, code and collections
36
+ tools are grouped. `tools/index.ts` is the advertised inventory.
37
+
38
+ Normally omit `session_id`: tools that support it resolve the current session.
39
+ Use an explicit ID only when supplied by the host or required by the task.
93
40
  `log_session_concepts`, `upsert_concepts`, `get_session_quiz_plan`,
94
- `record_attempt` and `get_gate_status` take it, optionally.
95
- `get_learner_profile`, `get_concept_graph`, `get_config` and `set_config` have
96
- no such argument at all. **Omit it everywhere.** The server resolves the
97
- current session itself, which is what lets the subagent's answers count toward
98
- the same gate. Only pass one if a hook handed you an id.
99
-
100
- ## The facts that have drifted before
101
-
102
- Check every one of these against the code when you touch a skill.
103
-
104
- - **`focus` defaults to `concept`, not `project`** (`DEFAULT_CONFIG` in
105
- `mcp/src/config.ts`). This was wrong in three skills at once. Grep before you
106
- write it: `grep -rn 'focus' skills/ user-skill/ agents/`.
107
- - **The `interleaved` one-question cap has exemptions.** In
108
- `mcp/src/tools/get_session_quiz_plan.ts`:
109
- `capped = cadence === 'interleaved' && !quiz.enforced && !explicitTopic`,
110
- and then `max = args.max ?? (capped ? 1 : max_questions_per_task)`.
111
- So `quiz.enforced` is exempt, an explicit `domain` or `slugs` is exempt, and
112
- an explicit `max` wins outright because it is read first. Read that code
113
- rather than trusting prose about it — including this paragraph.
114
- - **The Stop hook blocks when unenforced too.** `mcp/src/hooks/stop-quiz-check.ts`
115
- blocks either way; what `quiz.enforced` changes is that it skips the
116
- `min_minutes_between_quizzes` cooldown, takes the whole remaining budget
117
- instead of one question, and gates commits. Do not write "it never
118
- interrupts unless enforced".
119
- - **`get_learner_profile`'s lists are capped, and `known` is ordered by score.**
120
- `LIST_CAP = 8` covers `weak`, `due_for_review`, `projects`,
121
- `recent_concepts` and `skipped`; `KNOWN_CAP = 30` covers `known`, with the
122
- real count in `known_total`. `known` is sorted strongest-first, **not** by
123
- date. `get_concept_graph` caps at 200 nodes and sets `truncated`.
124
- - **The seed catalogue holds 87 concepts today** — 33 `web-auth`, 19 `git`, 18
125
- `react`, 17 `node-backend`, in `mcp/src/seed/*.json`. If you state that
126
- number in a skill, recount it first; a seed file gains concepts and the
127
- sentence does not.
128
-
129
- ## The dials are in the status bar, not above the stem
130
-
131
- Until 1.14 every plan item carried `ask_header` and the tutor printed it above
132
- the question: `[mode: ambient · focus: concept · level: easy · tier: 2
133
- mechanism]` — in the vocabulary of the day, when a single `mode` dial still
134
- existed. It existed for a real reason — on `concept` focus a deliberately
135
- transferable question reads as a vague one, and on `easy` a tier-2 question
136
- reads as shallow rather than as a runway — but it spent four settings' worth of
137
- screen above *every* stem to say something that is true for the whole session.
138
-
139
- Ambient state belongs somewhere ambient. `statusLine` in `mcp/src/statusline.ts`
140
- composes `[EKLAVYA concept · interleaved · easy]` for `eklavya statusline`,
141
- which the host's status bar runs — with `enforced` prepended only when it is
142
- set, since a segment that is always there is a segment nobody reads. `askHeader` is deleted,
143
- the plan no longer carries `ask_header`, and both hooks now say *ask the stem on
144
- its own*.
145
-
146
- Two consequences for anyone editing a skill:
147
-
148
- - **Never tell the model to compose a settings line.** Not the dials, not the
149
- tier, not `question: 2 of 3`. A line a skill assembles is a line the server
150
- cannot keep consistent, which is the reason this was centralised in the first
151
- place.
152
- - **`stripAskHeader` stays, and must.** Every attempt recorded while the line
153
- existed still has it inside the stem, and `questionFingerprint` (`store.ts`)
154
- hashes that text. Delete the stripper and the entire back catalogue changes
155
- fingerprint at once, so *never the same question twice* breaks for every
156
- question ever asked. It is now a guard for history, plus a model that invents
157
- a line anyway — plus the one thing that is still allowed above a stem, below.
158
-
159
- ### The one exception: who is asking
160
-
161
- The status bar carries the dials, and the `header` chip carries the
162
- attribution. Both are terminal paint. Claude Desktop draws a question card with
163
- no chip in it and has no status bar to run `eklavya statusline` in, so a
164
- question there arrived signed by nobody — which is the thing the chip existed
165
- to prevent, failing silently on a whole host.
166
-
167
- So `[Eklavya]` goes back above the stem, **on those hosts only**, and only that.
168
- `needsInlineAttribution` in `mcp/src/surface.ts` decides, `attributionRule`
169
- composes the sentence, and the plan returns it as `ask_attribution`. That field
170
- is why this is not a re-run of `ask_header`: the rule is composed once on the
171
- server, where the host is visible, rather than assembled by a skill that cannot
172
- see one.
173
-
174
- A third consequence follows from the two above:
175
-
176
- - **A skill file must not state the rule itself.** `writing-mcq.md` and
177
- `SKILL.md` point at `ask_attribution` and stop. A skill is static and a host
178
- is not, so a skill that spells out "set `header` and nothing else" is correct
179
- in a terminal and wrong in Claude Desktop, with no way to tell which it is
180
- being read in.
181
-
182
- The tier is deliberately nowhere on screen. A status bar refreshes on the host's
183
- cadence, so a tier there would sometimes name the previous question's
184
- difficulty, and a stale readout is worse than none. `level` covers what the tier
185
- was explaining: `easy` already means tiers 1-2.
186
-
187
- ## The tutor skill is an entry point plus references
188
-
189
- `skills/tutor/SKILL.md` was 5,296 words in one file, loaded whole whenever the
190
- model decided a task was non-trivial. It is now the part that decides *whether
191
- to act* — the log loop, checkpoint versus sweep, the shared budget, the tier
192
- ladder, the plan's authoritative fields, and a Red Flags table of the
193
- rationalizations that have each shipped a worse session — with the craft in
194
- three siblings:
195
-
196
- | File | Holds |
41
+ `record_attempt`, `get_gate_status`, `get_config` and `set_config` accept it for
42
+ session resolution. `memory_timeline` also accepts it, but as a filter: omitting
43
+ it shows the whole project. Check each schema instead of assuming all tools
44
+ accept the same fields.
45
+
46
+ Memory follows “search, choose, then get”. Search returns a small index;
47
+ `memory_get` returns the chosen narrative. Hydrating every result defeats the
48
+ context-saving design.
49
+
50
+ ## Preserve the tutoring contract
51
+
52
+ - Read defaults from `mcp/src/config.ts`: focus is `concept`. Search every skill,
53
+ user skill, agent and manual page when changing a shared setting.
54
+ - Planner output is authoritative. Interleaved questions are capped at one only
55
+ when unenforced and without an explicit topic; explicit `max` wins. Unenforced
56
+ Stop hooks can still ask questions.
57
+ - Settings belong in `eklavya statusline`, not a hand-composed question header.
58
+ Follow the plan's `ask_attribution` for host-specific `[Eklavya]` attribution;
59
+ never hardcode a terminal-only rendering rule. Do not display tier/counter
60
+ lines above a stem.
61
+ - `stripAskHeader` in `ask.ts` remains necessary for old recorded questions and
62
+ stable fingerprints. Removing it changes duplicate detection for history.
63
+ - Honor list caps and truncation. Profile `known` is strongest-first, not most
64
+ recent; graph results can be incomplete. Recount seed totals before quoting.
65
+ - Keep `declined` distinct from `dont_know`, MCQ grading capped as the runtime
66
+ requires, and one-question/resume behavior intact.
67
+
68
+ The tutor entry point chooses whether to act and points to required references:
69
+
70
+ | Reference | Read before |
197
71
  |---|---|
198
- | `references/writing-mcq.md` | the four-option shape, `answer_position`, distractors, plain language, the second question about a concept, `prereqs_unmet`, and how to record a stem |
199
- | `references/grading.md` | both scales, the mcq cap, feedback length, the four-step sequence a blank earns, `already_taught` |
200
- | `references/focus-and-level.md` | the three focuses, the earned level bands, the cadence contract, the enforced-mode gate retry |
201
-
202
- Two rules keep that split working.
203
-
204
- **Mark a reference REQUIRED at the point of use, never as an `@`-link.** An
205
- `@`-path is resolved eagerly by the host, which pulls the whole file into
206
- context and undoes the split. "Read `references/grading.md` before you grade",
207
- written where grading comes up, is what makes the model open it exactly when it
208
- needs it.
209
-
210
- **The entry point does not grow back, and the pointers stay honest.**
211
- `test/packaging.test.ts` fails above 2,000 whitespace tokens, and asserts the
212
- set of files named in SKILL.md is *equal* to the set on disk. Both directions
213
- matter. A pointer with no file is the worse half — the model is told the rules
214
- are elsewhere, cannot find them, and improvises, while nothing errors. A file
215
- with no pointer is the quieter half: it ships, `export-rules` inlines it, and
216
- Claude Code is never told to read it, so the same pedagogy differs by surface.
217
- An earlier version of that test harvested pointers from SKILL.md *and*
218
- `agents/tutor.md` into one list and asserted the list was non-empty — which
219
- passed with no pointers in SKILL.md at all, the exact state it was written to
220
- catch.
221
-
222
- ## Match the form to the failure
223
-
224
- Two failures need opposite wording, and using the wrong form measurably makes
225
- things worse. superpowers A/B tested this on their own dispatch-prompt
226
- guidance: the "don't do X" version produced **more** of the unwanted content
227
- than the "here is the shape" version — the distributions fully separated — and
228
- it did worse than giving no guidance at all.
229
-
230
- - **The model knows the rule and breaks it under pressure.** Discipline. Ban
231
- it, and name the excuse next to it: that is what the Red Flags table at the
232
- top of `SKILL.md` is, and what the shared budget, one-question and
233
- spent-question rules live in.
234
- - **The model complies and produces the wrong shape.** Craft. Bans backfire
235
- here. Describe the shape you want, in build order, and let the prohibitions
236
- fall out of it as properties of the finished thing.
237
-
238
- Writing a good multiple-choice question is the second kind, and
239
- `references/writing-mcq.md` was written as the first kind — *never restate the
240
- answer, no double negatives, avoid "which is NOT", do not number the options*.
241
- It is now a six-part recipe in build order followed by a checklist of
242
- properties, so the same rules arrive as "the answer appears among the options
243
- and nowhere in the stem" rather than as separate bans to weigh.
244
-
245
- **No nuance clauses in the recipe.** superpowers measured this separately: one
246
- appended "unless it matters" turns a reliable recipe into a noisy one, because
247
- it reopens the negotiation the recipe had settled. `writing-mcq.md` carried
248
- exactly one — *"Save the precise term for when the precision is the point"* —
249
- and it is gone. If an exception is real, it belongs in the plan's `framing`,
250
- which is server-side and authoritative, not in a hedge the model gets to weigh.
251
-
252
- `references/grading.md` keeps its prohibitions on purpose. Inflating a grade
253
- and offering to stop because someone is blanking are pressure failures, not
254
- shape failures: the model knows what honest grading is.
255
-
256
- ## The tutor skill has two readers
257
-
258
- `mcp/scripts/copy-assets.mjs` bundles the whole `skills/tutor/` directory to
259
- `mcp/dist/assets/tutor/`, and `eklavya export-rules` (`mcp/src/cli.ts`) strips
260
- the frontmatter and wraps it as a Cursor rules file. So the pedagogy is
261
- consumed by two editors.
262
-
263
- **Cursor has no progressive disclosure**, and that is the reason `export-rules`
264
- concatenates SKILL.md with every `references/*.md` in alphabetical order and
265
- says so in its preamble. A rules file is one document with `alwaysApply: true`,
266
- so "read `references/grading.md`" there is a pointer to nothing. Had the split
267
- shipped without the inlining, Cursor would have got the dispatch logic and none
268
- of the craft — and every test would still have passed. `test/cli.test.ts` now
269
- asserts a line from each reference reaches the output.
270
-
271
- Alphabetical rather than a hand-kept order: in an always-apply document the
272
- whole thing is in context at once, so order carries no meaning, and a listed
273
- order is one more place a new reference gets forgotten.
274
-
275
- **A missing reference is a hard failure there, not a warning.** `export-rules`
276
- reads the pointers out of SKILL.md and refuses to emit anything if one of them
277
- did not bundle, naming the file. It has to: the preamble promises the material
278
- is further down the document, so a half-bundled export is worse than none — the
279
- model is assured the rules are present and hunts for them instead of falling
280
- back on what it has. `copy-assets.mjs` only warns when a copy fails, so that
281
- state is reachable rather than hypothetical.
282
-
283
- Consequence of the two readers: **no Claude-Code-only instructions in any of
284
- the four files.** Slash-command names, plugin paths and hook mechanics belong
285
- in the command skills, not in the pedagogy. `AskUserQuestion` is the one
286
- unavoidable exception, and `agents/tutor.md` already carries the fallback for
287
- renderers that lack it.
288
-
289
- ## Consistency
290
-
291
- The same behaviour described in two skills has drifted apart before — that is
292
- how `focus: project` got into three files. The dials appear in
293
- `skills/mode/SKILL.md` and `user-skill/eklavya/SKILL.md`; the level bands
294
- appear in `skills/level/SKILL.md` and `tutor/references/focus-and-level.md`;
295
- the cadence cap appears in `mode`, `quiz` and that same reference. **When you
296
- change one, grep the others for the same claim.** Where any of them disagrees with the code,
297
- `mcp/src/config.ts` and the tool file are right and the skill is wrong.
72
+ | `tutor/references/writing-mcq.md` | Writing the stem, options and recorded question |
73
+ | `tutor/references/grading.md` | Grading, feedback, blanks and repeated teaching |
74
+ | `tutor/references/focus-and-level.md` | Applying focus, earned levels, cadence and gate retries |
75
+
76
+ Mark references REQUIRED at their point of use. Do not use eager `@` imports.
77
+ Keep the entry point within the packaging test's 2,000-whitespace-token limit.
78
+ Every referenced file must exist and every reference file must be reachable from
79
+ the entry point; the packaging test checks both directions.
80
+
81
+ For craft, describe the desired shape in build order, followed by a checklist.
82
+ For pressure failures such as grade inflation, retain explicit prohibitions and
83
+ the rationalization they prevent. Avoid vague exceptions that let the model
84
+ renegotiate a recipe; genuine framing exceptions belong in planner output.
85
+
86
+ ## Bundling and host limits
87
+
88
+ `copy-assets.mjs` bundles the tutor directory. `eklavya export-rules` strips
89
+ frontmatter and inlines reference files alphabetically for consumers that do
90
+ not load them on demand. Missing references must fail export, not emit partial
91
+ pedagogy. Keep shared tutor material independent of plugin paths, hook mechanics
92
+ and slash commands; the tutor agent supplies the plain-text fallback for
93
+ `AskUserQuestion`.
94
+
95
+ Rule export does not provide another editor with Claude Code hooks. Do not
96
+ advertise equivalent ambient learning on a host without that integration.
97
+ Delegation policy lives in `docs/subagent-policy.md`; the tutor exemption from
98
+ the implementer's “do not ask” directive must remain deliberate.
99
+
100
+ ## Documentation and checks
101
+
102
+ Update the corresponding manual page in the same PR: commands in `commands`,
103
+ settings in `dials`/`configuration`, grading in `grading-engine`, levels in
104
+ `levels-and-tiers`, artifacts in `dashboard`, and setup in `first-run`.
105
+ Update the landing command list if the public command set changes. Search other
106
+ skills for duplicated claims rather than fixing only the entry you touched.
107
+
108
+ Run the relevant packaging, CLI and tool tests from `mcp/` through `npm test`
109
+ so bundled assets are rebuilt. Tutor behavior changes also require before/after
110
+ evaluation (`eval/README.md`) and the live checkpoint acceptance check
111
+ (`CONTRIBUTING.md`). A successful build alone cannot verify prompt behavior.
112
+ Build `web/` for documentation edits and record any check not run.
@@ -168,6 +168,13 @@ tool exists to prevent. **Read `references/grading.md` before you grade** — bo
168
168
  scales, how long feedback may be, the sequence a blank earns, and what
169
169
  `already_taught` changes.
170
170
 
171
+ **A missed answer can come back with `explain`.** That is `explain_on_wrong` at
172
+ work: follow its `instruction` exactly. The page is written in the background
173
+ and opens by itself, so give the verdict, say the page is on its way, and go
174
+ back to the task — no longer explanation, no waiting. When the developer asks
175
+ for something to be explained as a page ("explain this to me", "make me a page
176
+ on this"), start the same background explainer whatever the setting says.
177
+
171
178
  ## The dials
172
179
 
173
180
  **Mode** is how hard to push, **focus** is what to teach, **cadence** is when to
@@ -110,6 +110,10 @@ export const setConfig = {
110
110
  max_new_concepts_per_session: z.number().int().min(0).max(50).optional(),
111
111
  max_stop_blocks_per_session: z.number().int().min(0).max(20).optional(),
112
112
  quiet: z.boolean().optional(),
113
+ explain_on_wrong: z
114
+ .boolean()
115
+ .optional()
116
+ .describe('Whether a missed question also gets an explainer page, written in the background and opened (default false). Usually set per project.'),
113
117
  auto_update: z
114
118
  .boolean()
115
119
  .optional()