@skillstate/opencode 2.2.2 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/README.md +194 -142
  2. package/dist/feedback.d.ts +178 -0
  3. package/dist/feedback.d.ts.map +1 -0
  4. package/dist/feedback.js +235 -0
  5. package/dist/feedback.js.map +1 -0
  6. package/dist/index.d.ts +58 -2
  7. package/dist/index.d.ts.map +1 -1
  8. package/dist/index.js +48 -3
  9. package/dist/index.js.map +1 -1
  10. package/dist/mode.d.ts +97 -0
  11. package/dist/mode.d.ts.map +1 -0
  12. package/dist/mode.js +111 -0
  13. package/dist/mode.js.map +1 -0
  14. package/dist/opencode-adapter.d.ts +10 -46
  15. package/dist/opencode-adapter.d.ts.map +1 -1
  16. package/dist/opencode-adapter.js +1 -64
  17. package/dist/opencode-adapter.js.map +1 -1
  18. package/dist/paper-mode.d.ts +421 -0
  19. package/dist/paper-mode.d.ts.map +1 -0
  20. package/dist/paper-mode.js +445 -0
  21. package/dist/paper-mode.js.map +1 -0
  22. package/dist/plugin.d.ts +218 -76
  23. package/dist/plugin.d.ts.map +1 -1
  24. package/dist/plugin.js +867 -232
  25. package/dist/plugin.js.map +1 -1
  26. package/dist/response-sink.d.ts +208 -0
  27. package/dist/response-sink.d.ts.map +1 -0
  28. package/dist/response-sink.js +243 -0
  29. package/dist/response-sink.js.map +1 -0
  30. package/dist/runtime.d.ts +203 -0
  31. package/dist/runtime.d.ts.map +1 -0
  32. package/dist/runtime.js +332 -0
  33. package/dist/runtime.js.map +1 -0
  34. package/dist/session-registry.d.ts +148 -0
  35. package/dist/session-registry.d.ts.map +1 -0
  36. package/dist/session-registry.js +236 -0
  37. package/dist/session-registry.js.map +1 -0
  38. package/dist/spec-loader.d.ts +81 -0
  39. package/dist/spec-loader.d.ts.map +1 -0
  40. package/dist/spec-loader.js +163 -0
  41. package/dist/spec-loader.js.map +1 -0
  42. package/dist/state-store.d.ts +126 -0
  43. package/dist/state-store.d.ts.map +1 -0
  44. package/dist/state-store.js +186 -0
  45. package/dist/state-store.js.map +1 -0
  46. package/dist/step-boundary.d.ts +91 -0
  47. package/dist/step-boundary.d.ts.map +1 -0
  48. package/dist/step-boundary.js +109 -0
  49. package/dist/step-boundary.js.map +1 -0
  50. package/dist/system-hint.d.ts +129 -0
  51. package/dist/system-hint.d.ts.map +1 -0
  52. package/dist/system-hint.js +166 -0
  53. package/dist/system-hint.js.map +1 -0
  54. package/dist/tools.d.ts +148 -0
  55. package/dist/tools.d.ts.map +1 -0
  56. package/dist/tools.js +350 -0
  57. package/dist/tools.js.map +1 -0
  58. package/package.json +3 -2
  59. package/dist/plugin-types.d.ts +0 -71
  60. package/dist/plugin-types.d.ts.map +0 -1
  61. package/dist/plugin-types.js +0 -8
  62. package/dist/plugin-types.js.map +0 -1
package/dist/plugin.js CHANGED
@@ -1,279 +1,914 @@
1
1
  /**
2
- * Static OpenCode plugin — the SINGLE SOURCE OF TRUTH for the skillstate
3
- * host integration. `OpenCodeAdapter.generatePluginCode` emits a thin loader
4
- * that imports `createSkillStatePlugin` from this module; the per-project
5
- * state resolution lives in `@skillstate/core`
6
- * (`resolveHostStateForCwd`, re-exported here) and the hook logic
7
- * (envelope read/write, ⊕ merge, patch extraction) in the core
8
- * hook-runtime — this module only adapts it to the OpenCode hooks.
9
- *
10
- * Hooks (opencode 1.17 contract, verified on host):
11
- * - `experimental.chat.messages.transform` — entries are `{ info, parts }`
12
- * envelopes (role on `info.role`); the pipeline keeps the ORIGINAL array
13
- * reference, so trimming mutates in place; the state is injected as a
14
- * synthetic `{ info, parts }` element. Real O(1) prompt footprint.
15
- * - `experimental.session.compacting` — pushes the state into
16
- * `output.context` so the compaction summary preserves it.
17
- * - `tool.execute.after` — the tool response is `output.output`; a fenced
18
- * ```json `state_patch` block is merged (paper ⊕: null deletes) and saved.
19
- *
20
- * AGENT-SCOPED STATE: the opencode hook inputs carry the session id
21
- * (`input.sessionID`; message envelopes carry `info.sessionID`). The MAIN
22
- * session resolves to the ROOT state file
23
- * `<cwd>/.skillstate/skillstate.json` — the same file the skillstate MCP
24
- * tools and the CLI address, so the injected state and `state.patch` can
25
- * never disagree. SUB-AGENT sessions (registered from the host event bus —
26
- * `session.created`/`session.updated` carry `info.parentID`) resolve to
27
- * isolated `agents/<parentPrefix>-<sessionPrefix>/` copies and never
28
- * last-writer-win the main state; the main agent folds them back with
29
- * `agent.merge`. Writes go through the core cross-process sync lock
30
- * (`lockStateWrite`) so a state file is never interleaved between
31
- * processes.
2
+ * `@skillstate/opencode` — the OpenCode **v2** plugin.
3
+ *
4
+ * ── What this replaces ───────────────────────────────────────────────────
5
+ *
6
+ * The v1 integration rewrote the conversation on every model request. It
7
+ * kept the system messages and the last three non-system messages, dropped
8
+ * everything else from `output.messages`, and appended a synthetic
9
+ * `role: "user"` message containing the raw state JSON. The reported
10
+ * failure was that the agent stopped doing the user's task and started
11
+ * emitting state JSON instead.
12
+ *
13
+ * Both halves of that were destructive, and neither was a model quirk:
14
+ *
15
+ * 1. The injected message landed LAST, so for the model it was the current
16
+ * instruction — it displaced the user's actual request.
17
+ * 2. `slice(-3)` deleted the task statement, the tool results and the
18
+ * errors the agent had just been handed. It was reasoning about work it
19
+ * could no longer see.
20
+ *
21
+ * The MCP server made it worse: `spec.get` returned a procedural spec whose
22
+ * default was `INTERCODE_CTF_SPEC`, whose instructions read "You are an
23
+ * autonomous CTF agent ... hidden flag somewhere on its filesystem". A
24
+ * model told to look for a flag looks for a flag. (Fixed: the default is now
25
+ * the neutral `GENERIC_PROCEDURE_SPEC`, and its instructions describe the
26
+ * storage format instead of prescribing a way of working.)
27
+ *
28
+ * ── The v2 design ────────────────────────────────────────────────────────
29
+ *
30
+ * Two modes, each enforced by a test. They are different contracts with the
31
+ * model, not variants of one behaviour:
32
+ *
33
+ * - **`notes` (default).** Contribute one additive, bounded fragment to
34
+ * `event.system` and leave the transcript alone. The agent sees its own
35
+ * history the way the host intends, and the saved notes ride alongside it.
36
+ * This is the mode that fixed the v1 failure, and it is the default for
37
+ * exactly that reason.
38
+ * - **`paper` (opt-in).** Replace the model-facing context with
39
+ * Aₜ = (P, Σₜ, Oₜ) — the paper's Appendix A.4 prompt, byte-verbatim — and
40
+ * apply the `state_patch` the model emits in response. This is the paper's
41
+ * specification, and it is a real behavioural change: the model stops
42
+ * seeing its transcript, because §3.2 discards the reasoning trace by
43
+ * construction. Select it with `mode: "paper"` in the project's
44
+ * `skillstate.json` or `SKILLSTATE_MODE=paper`; see `mode.ts`.
45
+ *
46
+ * The default is `notes` and must stay that way: a default that discards the
47
+ * user's task is the v1 bug under a new name.
48
+ *
49
+ * Three rules, each enforced by a test:
50
+ *
51
+ * - **Notes mode never mutates `event.messages`.** The plugin contributes
52
+ * one additive fragment to `event.system` and leaves the transcript alone.
53
+ * See `tests/opencode/context-integrity.test.ts`.
54
+ * - **Never inject behavioural instructions in notes mode.** The system
55
+ * fragment describes what the notes are and when to use them; it contains
56
+ * no "you must", no "always", and no output format. See
57
+ * `system-hint.ts`.
58
+ * - **Inert until used.** A project with no state file gets no system
59
+ * fragment at all and behaves exactly like vanilla OpenCode. No files are
60
+ * created by loading the plugin.
61
+ *
62
+ * ── Native tools AND the MCP server, on purpose ──────────────────────────
63
+ *
64
+ * This package does not replace `@skillstate/mcp`; it sits beside it.
65
+ *
66
+ * - The native tools ({@link registerTools}) are the fast path inside
67
+ * opencode: a typed schema, structured output, no JSON-RPC round-trip and
68
+ * no untyped text result.
69
+ * - The MCP server is the portable path. It is what every other
70
+ * MCP-capable host reads, and the only way to reach this state from a
71
+ * client that is not opencode.
72
+ *
73
+ * Both address the same `<project>/.skillstate/skillstate.json`, so they
74
+ * cannot disagree about what is saved. `skillstate init` registers both.
75
+ *
76
+ * The reason v1 needed the MCP server is gone: an opencode v1 plugin could
77
+ * not contribute first-class tools at all.
78
+ *
79
+ * Load it from `opencode.json(c)`:
80
+ *
81
+ * ```json
82
+ * { "plugins": ["@skillstate/opencode"] }
83
+ * ```
32
84
  */
85
+ import { Plugin } from '@opencode/plugin';
33
86
  import * as fs from 'node:fs';
34
- import * as os from 'node:os';
35
87
  import * as path from 'node:path';
36
- import { findFencedPatch, lockStateWrite, mergePatch, readStateEnvelope, resolveAgentIdFromSession, resolveHostStateForCwd, saveStateEnvelope, } from '@skillstate/core';
37
- export * from './plugin-types.js';
88
+ import { fileURLToPath } from 'node:url';
89
+ import { resolvePluginMode } from './mode.js';
90
+ import { HOST_ACTION_NOTE, applyPaperContext, buildPaperPrompt, latestObservation } from './paper-mode.js';
91
+ import { FeedbackQueue } from './feedback.js';
92
+ import { PaperStateSink, isTextEnded } from './response-sink.js';
93
+ import { SessionRegistry, stateScopeFor } from './session-registry.js';
94
+ import { SpecResolver } from './spec-loader.js';
95
+ import { DEFAULT_MAX_STEPS, INVALID_PATCH, RuntimeDriver } from './runtime.js';
96
+ import { StepBoundary } from './step-boundary.js';
97
+ import { ProjectStateStore } from './state-store.js';
98
+ import { buildStateHint, driftNotice } from './system-hint.js';
99
+ import { registerTools } from './tools.js';
100
+ /** Stable plugin id — scopes plugin storage and identifies it in `/api/plugin`. */
101
+ export const PLUGIN_ID = 'skillstate';
38
102
  /**
39
- * Resolve the per-project state file for a session working directory
40
- * (`cwd` of the current opencode session) — the core single source of
41
- * truth (`resolveHostStateForCwd`): `<cwd>/.skillstate/skillstate.json`,
42
- * or the global bucket `<home>/.skillstate/global/skillstate.json` when
43
- * cwd equals home. A non-empty `agentId` scopes the file under
44
- * `<bucket>/agents/<agentId>/skillstate.json`. Pure path arithmetic via
45
- * `path.resolve`, no filesystem access.
103
+ * The action carried forward when a turn produced no usable patch.
104
+ *
105
+ * NOT `__invalid_patch__`, though the paper names that sentinel at §5.1 line
106
+ * 9. It is a return value there — what the step function hands back to signal
107
+ * that Σ is unchanged — and forwarding it into the prompt as the next action
108
+ * is meaningless to a model: it is a name, not a request. The retry instruction
109
+ * the model actually needs already rides in Oₜ through the feedback queue, so
110
+ * this only has to say "keep going", and the queue says why.
46
111
  */
47
- export { resolveHostStateForCwd as resolveStatePathForCwd };
48
- export { mergePatch };
112
+ export const CONTINUE_ACTION = 'continue';
49
113
  /**
50
- * Read the state file. Missing or corrupt files yield `{}` (best-effort).
51
- * The on-disk envelope is `{ version: 1, state }` (migrations-compatible);
52
- * a bare object is tolerated and treated as the state itself. Thin fs
53
- * adapter over the core hook-runtime {@link readStateEnvelope}.
114
+ * The action the model last asked for, per session, waiting for the turn to end.
115
+ *
116
+ * A text block ending is not a turn ending — a model that narrates and then
117
+ * calls a tool ends a block and is nowhere near done. The host says when the
118
+ * turn is actually over, and that is the only point at which asking for the
119
+ * next step is correct.
54
120
  */
55
- export function readSkillState(statePath) {
56
- return readStateEnvelope(statePath, (p) => fs.readFileSync(p, 'utf-8'));
57
- }
121
+ const lastAction = new Map();
58
122
  /**
59
- * Persist the state file (best-effort: read-only environments are ignored).
60
- * Creates the parent directory when missing (the per-project resolver may
61
- * target a fresh `<cwd>/.skillstate/agents/<id>/`). Writes the
62
- * `{ version: 1, state }` envelope so `migrate()`/runtime resume read the
63
- * same file — via the core hook-runtime {@link saveStateEnvelope} — under
64
- * the cross-process sync lock {@link lockStateWrite} (2-3 parallel agent
65
- * processes never interleave state writes).
123
+ * Sessions whose step spent all `k + 1` attempts, for §6.4's synthetic
124
+ * observation. Session-scoped rather than global because the attempt budget is
125
+ * per session: one session stalling must not put an invalidation in front of
126
+ * another's next prompt.
66
127
  */
67
- export function saveSkillState(statePath, state) {
128
+ const invalidations = new Map();
129
+ /**
130
+ * Whether an event says the host has finished a step.
131
+ *
132
+ * `session.step.ended`, measured — not `session.idle`, which is what the SDK
133
+ * type reads like and which the host never emits. Recorded every event type the
134
+ * plugin receives for one run: 2x `session.step.ended`, 2x `session.text.ended`,
135
+ * and zero of `session.idle`. So the trigger that was supposed to turn the loop
136
+ * never fired once, and the loop could not turn. The SDK exports
137
+ * `SessionMessageIdle`, which is a different thing entirely and reads like an
138
+ * event name because it is not one.
139
+ */
140
+ /**
141
+ * Write `.skillstate/.build.json`: the plugin id, the package version, and the
142
+ * build's own mtime and size.
143
+ *
144
+ * Deliberately cheap and deliberately once. The mtime is the whole point — it is
145
+ * the thing that differs between two builds of identical source, which is exactly
146
+ * the case a version string cannot see.
147
+ *
148
+ * Never throws and never blocks setup: a stamp that fails to write is a missing
149
+ * field in a diagnostic file, and a plugin that cannot start is worse.
150
+ */
151
+ function writeBuildStamp(directory) {
68
152
  try {
69
- fs.mkdirSync(path.dirname(statePath), { recursive: true });
70
- lockStateWrite(statePath, fs, () => saveStateEnvelope(statePath, state, (p, data) => fs.writeFileSync(p, data)));
153
+ // The module's OWN file, not a guessed sibling. The first version of this
154
+ // stat'd `path.join(here, 'index.js')` — right under `dist/`, absent under
155
+ // `src/` — so the try/catch swallowed ENOENT and the function was silently
156
+ // dead in every test that ran it. The same shape as this whole day: guessing
157
+ // at a path instead of measuring the thing already in hand.
158
+ const entry = fileURLToPath(import.meta.url);
159
+ const here = path.dirname(entry);
160
+ const stat = fs.statSync(entry);
161
+ // No inner try: the outer one already covers this, and a stamp that cannot be
162
+ // written is a missing diagnostic, not a failure. The inner catch returned
163
+ // 'unknown', which is a number no run could ever check against anything — an
164
+ // unreachvable branch that the coverage gate was right to refuse.
165
+ // `version` is optional in a package.json, so it may be absent, and absence
166
+ // is the honest rendering: JSON.stringify drops an undefined field on its
167
+ // own. `?? 'unknown'` produced a string no run could check against anything,
168
+ // and a ternary guarding the field produced a second unreachable branch for
169
+ // the coverage gate to refuse. Both are the same mistake — a fallback value
170
+ // where a missing one was already correct.
171
+ const version = JSON.parse(fs.readFileSync(path.join(here, '..', 'package.json'), 'utf-8')).version;
172
+ // Only into an EXISTING `.skillstate/`. The plugin is inert for a project
173
+ // that never ran `skillstate init` — and it uses the presence of that
174
+ // directory as the definition of "initialised". Creating it here would
175
+ // initialise every project the plugin is installed into, which is not a
176
+ // diagnostic, it is a side effect with consequences.
177
+ //
178
+ // Caught by an existing test, "creates no files in a project that has no
179
+ // state", three lines after this was written. A run with a state file is a
180
+ // run worth stamping; a project that has none is not a run at all.
181
+ const dir = path.join(directory, '.skillstate');
182
+ if (!fs.existsSync(dir))
183
+ return;
184
+ fs.writeFileSync(path.join(dir, '.build.json'), `${JSON.stringify({ plugin: PLUGIN_ID, version, distMtimeMs: stat.mtimeMs, distBytes: stat.size }, null, 2)}\n`);
71
185
  }
72
186
  catch {
73
- // Best-effort: read-only environments or permission issues.
187
+ // A diagnostic that cannot be written is not a reason to refuse to run.
74
188
  }
75
189
  }
76
190
  /**
77
- * Atomic READ-MERGE-WRITE of one `state_patch` (paper ⊕: null deletes):
78
- * the whole critical section runs inside {@link lockStateWrite}, so two
79
- * concurrent writers apply BOTH patches instead of racing between the
80
- * read and the write. Best-effort: lock contention or unwritable state
81
- * files are swallowed — the tool flow never breaks.
191
+ * Write `.skillstate/.run.json`: why the paper-mode loop last declined to spend
192
+ * a step, and the ceiling it was running under.
193
+ *
194
+ * The loop's end is otherwise invisible. `advance` returns `null` for a terminal
195
+ * action, a host refusal, a turn with no action, and the ceiling, and the
196
+ * caller cannot tell them from the return value — so a run stopped at the
197
+ * ceiling leaves a transcript that looks exactly like a run that finished. That
198
+ * is measured, not hypothetical: a 90-file run stopped at step 100 mid-file-79
199
+ * and was read as a model that lost track of its running sum at file 78.
200
+ *
201
+ * Same rules as the build stamp: only into an existing `.skillstate/`, and never
202
+ * allowed to throw. A run record that cannot be written is a missing diagnostic,
203
+ * not a reason to take the session down.
82
204
  */
83
- export function mergeSkillState(statePath, patch) {
205
+ function writeRunRecord(directory, stop, maxSteps) {
84
206
  try {
85
- fs.mkdirSync(path.dirname(statePath), { recursive: true });
86
- let merged = {};
87
- lockStateWrite(statePath, fs, () => {
88
- merged = mergePatch(readSkillState(statePath), patch);
89
- saveStateEnvelope(statePath, merged, (p, data) => fs.writeFileSync(p, data));
90
- });
91
- return merged;
207
+ const dir = path.join(directory, '.skillstate');
208
+ if (!fs.existsSync(dir))
209
+ return;
210
+ fs.writeFileSync(path.join(dir, '.run.json'), `${JSON.stringify({ stop, maxSteps }, null, 2)}\n`);
92
211
  }
93
212
  catch {
94
- return readSkillState(statePath);
213
+ // A diagnostic that cannot be written is not a reason to refuse to run. A
214
+ // `.skillstate` that is a FILE rather than a directory passes the
215
+ // `existsSync` above and fails the write here, and a run must survive that.
95
216
  }
96
217
  }
218
+ function isStepEnded(event) {
219
+ if (typeof event !== 'object' || event === null)
220
+ return false;
221
+ const typed = event;
222
+ return (typed.type === 'session.step.ended' &&
223
+ typeof typed.data?.sessionID === 'string');
224
+ }
97
225
  /**
98
- * Extract the `state_patch` object from an LLM response's fenced ```json
99
- * block; `null` when there is no block, it is malformed, or it carries no
100
- * object-shaped `state_patch`. Thin adapter over the core hook-runtime
101
- * {@link findFencedPatch} (the invalid/truncated outcomes collapse to
102
- * `null`, preserving the legacy boolean contract).
226
+ * Record a failed step request, so "the host declined" is not a guess.
227
+ *
228
+ * @non-paper diagnostics. Enabled by `SKILLSTATE_DEBUG_PROMPT`, appended to
229
+ * the same file, tagged so it cannot be mistaken for a prompt record.
103
230
  */
104
- export function extractPatch(response) {
105
- const result = findFencedPatch(response);
106
- return 'patch' in result ? result.patch : null;
231
+ function recordPromptFailure(error) {
232
+ const path = process.env['SKILLSTATE_DEBUG_PROMPT'];
233
+ if (path === undefined || path.length === 0)
234
+ return;
235
+ const message = error instanceof Error ? `${error.name}: ${error.message}` : String(error);
236
+ try {
237
+ fs.appendFileSync(path, `${JSON.stringify({ promptFailure: message })}\n`);
238
+ }
239
+ catch {
240
+ // Diagnostics must never break the agent loop.
241
+ }
107
242
  }
108
243
  /**
109
- * Agent id for an opencode hook call.
110
- *
111
- * MAIN SESSION → `''` (the ROOT state file `<cwd>/.skillstate/skillstate.json`
112
- * — the SAME file the skillstate MCP tools and the CLI address, so the
113
- * injected state and `state.patch` can never disagree). A session
114
- * registered as a SUB-AGENT via the host event bus
115
- * (`session.created`/`updated` carry `info.parentID`) resolves to
116
- * `<parentPrefix>-<sessionPrefix>` — an isolated copy under `agents/` that
117
- * never last-writer-wins the main state; the main agent folds it back with
118
- * `agent.merge`. No session id at all → `''` (root: a single context).
244
+ * Record every event type the plugin actually receives.
245
+ *
246
+ * @non-paper diagnostics, same file. The advance is triggered by one event
247
+ * type and one, and a trigger that never fires is indistinguishable from one
248
+ * that is wired wrong — so the arrival counts have to be visible. Cheap, and
249
+ * it would have saved guessing.
119
250
  */
120
- export function pluginAgentId(input, messages) {
121
- const direct = resolveAgentIdFromSession(input?.sessionID);
122
- const sessionPrefix = direct.length > 0
123
- ? direct
124
- : resolveAgentIdFromSession((messages ?? []).find((m) => typeof m.info?.sessionID === 'string' &&
125
- m.info.sessionID.length > 0 &&
126
- m.info.sessionID !== 'skillstate')?.info.sessionID);
127
- if (sessionPrefix.length === 0)
128
- return '';
129
- return scopedAgentId(sessionPrefix);
251
+ export function recordEvent(path, type) {
252
+ if (path === undefined || path.length === 0)
253
+ return;
254
+ try {
255
+ fs.appendFileSync(path, `${JSON.stringify({ event: type })}\n`);
256
+ }
257
+ catch {
258
+ // Diagnostics must never break the agent loop.
259
+ }
130
260
  }
131
261
  /**
132
- * Widen an agent id for a registered sub-agent session:
133
- * `<parentPrefix>-<sessionPrefix>`. Plain sessions resolve to `''` (the
134
- * main/root scope — NOT their own agents/ copy).
262
+ * Whether the host has just executed a tool for this request.
263
+ *
264
+ * A tool result in the transcript is the observable edge of "an action ran".
265
+ * There is no event that says so in a shape this plugin can trust — and an
266
+ * event the host does not wait for is what caused the read-after-write race
267
+ * fixed in `response-sink.ts`, so the transcript is the more reliable of the
268
+ * two here as well as the more available one.
135
269
  */
136
- export function scopedAgentId(agentId) {
137
- const parent = SUB_AGENT_PARENTS.get(agentId);
138
- return parent === undefined ? '' : `${parent}-${agentId}`;
270
+ /** Tool-result parts in the newest tool message, counted. */
271
+ export function countToolResults(messages) {
272
+ const last = messages[messages.length - 1];
273
+ if (last === undefined || last.role !== 'tool')
274
+ return 0;
275
+ if (!Array.isArray(last.content))
276
+ return 0;
277
+ return last.content.filter((part) => typeof part === 'object' &&
278
+ part !== null &&
279
+ String(part.type).startsWith('tool-result')).length;
139
280
  }
281
+ /** Whether the newest message carries a tool result. One definition, one use. */
282
+ function hasToolResult(messages) {
283
+ return countToolResults(messages) > 0;
284
+ }
285
+ /** How much of an observation the diagnostic records. */
286
+ const DEBUG_OBSERVATION_CHARS = 400;
140
287
  /**
141
- * Record a session→parent edge from the host event stream. `sessionId`
142
- * with a non-empty `parentID` registers that session as a sub-agent of
143
- * `parentID`; an empty `parentID` (the main session being updated after
144
- * the fact) clears a stale registration. Exposed for tests.
288
+ * Append what the host actually handed us to a file, for diagnosis.
289
+ *
290
+ * @non-paper diagnostics. Enabled by `SKILLSTATE_DEBUG_PROMPT=<path>`.
291
+ *
292
+ * This exists because of a bug that was invisible from the inside for a
293
+ * long time. The model would run a tool, get the answer, and never record
294
+ * it — which looks exactly like a model refusing to cooperate, and sent the
295
+ * search through prompt slots, model choice and spec wording. The cause was
296
+ * the SHAPE: OpenCode v2 delivers a tool result as
297
+ * `{ type: 'tool-result', result: { value } }`, so a reader that only knew
298
+ * `{ type: 'text', text }` made Oₜ permanently empty without ever throwing.
299
+ *
300
+ * The dump records the part types alongside the extracted text, so that
301
+ * class of failure is visible on sight: a `tool-result` in the list next to
302
+ * an empty `observation` says the reader, not the model, is at fault.
303
+ *
304
+ * Append-only so a session's turns accumulate in order, and every failure
305
+ * is swallowed — diagnostics must never break the agent loop.
145
306
  */
146
- export function registerSessionParent(sessionId, parentId) {
147
- if (typeof sessionId !== 'string' || sessionId.length === 0)
148
- return;
149
- const session = resolveAgentIdFromSession(sessionId);
150
- if (session.length === 0)
307
+ export function dumpPromptShape(path, messages, state) {
308
+ if (path === undefined || path.length === 0)
151
309
  return;
152
- const parent = resolveAgentIdFromSession(parentId);
153
- if (parent.length === 0 || parent === session) {
154
- SUB_AGENT_PARENTS.delete(session);
310
+ const record = {
311
+ turn: messages.length,
312
+ // Σ as the model was shown it, not as it ended up on disk. A model that
313
+ // writes back a stale total is indistinguishable from one that was never
314
+ // given a fresh one, and those are opposite bugs.
315
+ state,
316
+ roles: messages.map((m) => m.role),
317
+ partTypes: messages.map((m) => Array.isArray(m.content)
318
+ ? m.content.map((part) => typeof part === 'object' && part !== null
319
+ ? String(part.type)
320
+ : typeof part)
321
+ : typeof m.content),
322
+ observation: latestObservation(messages).content.slice(0, DEBUG_OBSERVATION_CHARS),
323
+ };
324
+ try {
325
+ fs.appendFileSync(path, `${JSON.stringify(record)}\n`);
326
+ }
327
+ catch {
328
+ // Diagnostics must never break the agent loop.
329
+ }
330
+ }
331
+ /**
332
+ * One line per turn of the anti-drift diagnostic.
333
+ *
334
+ * The drift notice has a claim attached to it — "the model drifts, the notice
335
+ * brings it back" — and neither half can be checked from inside the process.
336
+ * A notice in a prompt is not an observation of a model: the fragment may be
337
+ * built correctly and the model may ignore it, and the two look identical
338
+ * from the code's side. Worse, both look identical from the *outside* too,
339
+ * which is what made the earlier `tool-result` bug so expensive to find.
340
+ *
341
+ * So each line carries the evidence that distinguishes them:
342
+ *
343
+ * - `notice` — was the drift sentence in the fragment that went out this turn;
344
+ * - `writes` — how many times the state file had changed when it went out, so
345
+ * a notice that repeats forever is visible as a flat counter;
346
+ * - `fragments` — how many turns had passed without a change.
347
+ *
348
+ * That is enough to say "the notice fired and the state moved afterwards" or
349
+ * "the notice fired and nothing happened", which is the only claim worth
350
+ * making about it.
351
+ *
352
+ * @non-paper diagnostics. Enabled by `SKILLSTATE_DEBUG_DRIFT=<path>`.
353
+ * Separate from {@link dumpPromptShape} because it answers a different
354
+ * question: that one asks what the host sent, this one asks what the model
355
+ * did about what we sent.
356
+ */
357
+ export function dumpDrift(path, record) {
358
+ if (path === undefined || path.length === 0)
155
359
  return;
360
+ try {
361
+ fs.appendFileSync(path, `${JSON.stringify(record)}\n`);
362
+ }
363
+ catch {
364
+ // Diagnostics must never break the agent loop.
156
365
  }
157
- SUB_AGENT_PARENTS.set(session, parent);
158
366
  }
159
- /** Test-only: forget every registered session→parent edge. */
160
- export function resetSessionParents() {
161
- SUB_AGENT_PARENTS.clear();
367
+ /**
368
+ * One line per step, for the run that answered correctly while its state
369
+ * under-reported the work. Enabled by `SKILLSTATE_DEBUG_STEPS=<path>`.
370
+ *
371
+ * The 30-file measurement produced the most confusing result in this project's
372
+ * history: paper mode answered correctly, its state ended 25/30, and the
373
+ * control's ended 30/30 complete. Twenty-five patches were emitted and all
374
+ * twenty-five landed, so no patch was lost — the model read every file and
375
+ * declined to patch the last five, while the loop kept driving it. From the
376
+ * outside that is indistinguishable from the loop stopping, from the model
377
+ * silently abandoning the protocol, and from the ceiling being hit.
378
+ *
379
+ * So this prints the thing that tells those apart: at every step, whether a
380
+ * patch was applied, what the state looked like, and whether the driver asked
381
+ * again. Every previous wrong guess in this file came from reasoning about the
382
+ * loop instead of watching it.
383
+ */
384
+ export function dumpStepTrace(path, record) {
385
+ if (path === undefined || path.length === 0)
386
+ return;
387
+ try {
388
+ fs.appendFileSync(path, `${JSON.stringify(record)}\n`);
389
+ }
390
+ catch {
391
+ // Diagnostics must never break the agent loop.
392
+ }
162
393
  }
163
- /** Synthetic message ids for the injected state carrier. */
164
- const STATE_MESSAGE_ID = 'skillstate-state-inject';
165
394
  /**
166
- * The session ids known to be SUB-AGENT sessions, keyed by session id →
167
- * parent session id. Populated from the `event` hook
168
- * (`session.created`/`session.updated` carry `info.parentID`); consulted
169
- * when resolving an agent id so a sub-agent's state lands in the SAME
170
- * agents/<parent>/<session-8>/ scope as its hook-session (the task tool
171
- * spawns sessions whose ids never appear as sub-agent prefixes — without
172
- * this map a sub-agent would silently write the MAIN state).
395
+ * The step ceiling, or `undefined` to keep the default.
396
+ *
397
+ * A malformed value is ignored rather than thrown on or silently clamped: a
398
+ * typo in an environment variable should leave the ceiling where the code says
399
+ * it is, not quietly become some other number that then gets measured.
173
400
  */
174
- const SUB_AGENT_PARENTS = new Map();
401
+ export function maxStepsFromEnv() {
402
+ const raw = process.env['SKILLSTATE_MAX_STEPS'];
403
+ if (raw === undefined || raw.length === 0)
404
+ return undefined;
405
+ // `Number`, not `parseInt`: parseInt('12.5') is 12, which is the silent
406
+ // truncation the comment above warns against — a ceiling set to a fifth more
407
+ // than asked for, measured and reported as if it were what was asked.
408
+ const parsed = Number(raw);
409
+ return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : undefined;
410
+ }
175
411
  /**
176
- * Build the OpenCode plugin function with the same behavior for every host
177
- * entry point (thin generated loaders, direct imports).
178
- *
179
- * State resolution is ALWAYS per-project: the state file path is computed
180
- * from the session cwd on EVERY hook call via
181
- * `resolveStatePathForCwd(process.cwd(), os.homedir(), agentId)` — each
182
- * project gets its own `<cwd>/.skillstate/`, each session (sub-agent) its
183
- * isolated `agents/<session>/` copy, and a session launched from `$HOME`
184
- * uses the global bucket.
412
+ * The plugin definition.
413
+ *
414
+ * `setup` wires the session registry, the project state store, the native
415
+ * tools, the mode resolver and the one `context` hook, then returns a cleanup
416
+ * function.
417
+ *
418
+ * - a {@link SessionRegistry}, fed by the server event stream, so a
419
+ * sub-agent session is recognised and given its own state file;
420
+ * - a {@link ProjectStateStore} rooted at the plugin's own project
421
+ * location, so two checkouts served by one OpenCode server never share
422
+ * state;
423
+ * - a {@link SpecResolver} for paper mode's P, so a project that ships its
424
+ * own `skill-spec.json` gets its own procedure;
425
+ * - a {@link PaperStateSink}, in paper mode only, which applies the
426
+ * `state_patch` the model emits;
427
+ * - native tools plus a single `context` hook whose body depends on the mode.
428
+ *
429
+ * The event subscription is the only resource the plugin owns, so the
430
+ * returned cleanup aborts it. Hook and tool registrations are disposed by
431
+ * OpenCode when the plugin unloads.
185
432
  */
186
- export function createSkillStatePlugin(options = {}) {
187
- const resolvePath = (agentId) => resolveHostStateForCwd(process.cwd(), os.homedir(), agentId);
188
- const maxHistory = options.maxHistoryMessages ?? 3;
189
- return async () => {
190
- return {
191
- // ── Session registry ──────────────────────────────────────────────
192
- // The host event bus carries full Session objects on
193
- // session.created/updated — including `parentID`. Registering here
194
- // is what makes sub-agent scoping work: a Task sub-agent's session
195
- // (parentID set) resolves to agents/<parent>-<session>/ BEFORE its
196
- // first hook fires, so it never touches the parent's state file.
197
- event: async ({ event }) => {
198
- const payload = event;
199
- if (payload === null ||
200
- typeof payload !== 'object' ||
201
- payload['type'] !== 'session.created' && payload['type'] !== 'session.updated') {
202
- return;
203
- }
204
- const info = payload['properties']?.['info'];
205
- if (info === null || typeof info !== 'object')
206
- return;
207
- const record = info;
208
- registerSessionParent(record['id'], record['parentID']);
209
- },
210
- // ── O(1) history trimming ──────────────────────────────────────────
211
- // Filters messages BEFORE each LLM call: keeps all system messages
212
- // plus the last `maxHistory` non-system messages, then injects a
213
- // synthetic state element. Old messages are DROPPED from the prompt,
214
- // not just hidden.
215
- 'experimental.chat.messages.transform': async (input, output) => {
216
- const agentId = pluginAgentId(input, output.messages);
217
- const state = readSkillState(resolvePath(agentId));
218
- const messages = output.messages;
219
- const systemMessages = messages.filter((m) => m.info.role === 'system');
220
- const trimmed = messages
221
- .filter((m) => m.info.role !== 'system')
222
- .slice(-maxHistory);
223
- // Synthetic state carrier — a `{ info, parts }` envelope whose text
224
- // part carries the current state JSON.
225
- const stateMessage = {
226
- info: {
227
- id: STATE_MESSAGE_ID,
228
- sessionID: 'skillstate',
229
- role: 'user',
230
- time: { created: 0 },
231
- agent: 'skillstate',
232
- model: { providerID: 'skillstate', modelID: 'skillstate' },
433
+ export const SkillStatePlugin = Plugin.define({
434
+ id: PLUGIN_ID,
435
+ async setup(ctx) {
436
+ const sessions = new SessionRegistry();
437
+ const scopeFor = (sessionID) => stateScopeFor(sessions, sessionID);
438
+ // `ctx.location.project.canonical` is the canonical checkout, stable
439
+ // across worktrees and symlinks. The v1 plugin used `process.cwd()`,
440
+ // which in v2 is the server's cwd, not the session's project.
441
+ const directory = ctx.location.project.canonical;
442
+ const store = new ProjectStateStore({ directory });
443
+ // Stamp the build, once, so a run records WHICH plugin it used.
444
+ //
445
+ // The host resolves plugins by workspace rather than by the name in
446
+ // `opencode.json` (measured: asking for a package that does not exist loads
447
+ // the one that does), and the plugin loads from `dist/`. So a run's
448
+ // behaviour depends on a build that is not named anywhere in its own output,
449
+ // and a fix that was committed without a rebuild produces a measurement of
450
+ // the previous version while looking like a measurement of this one.
451
+ //
452
+ // This project has already paid for that twice: an alias gap made the bench
453
+ // tests silently load a stale dist, and a fixture knob was measured after
454
+ // the stand had been seeded with an older fixture. Both were invisible.
455
+ void writeBuildStamp(directory);
456
+ const mode = resolvePluginMode({ directory }).mode;
457
+ const specs = new SpecResolver();
458
+ // Resolved in BOTH modes now, and the difference is what each one does
459
+ // with it. Paper mode formats P into the prompt and validates every patch
460
+ // against the schema; notes mode has no P to format and does not enforce,
461
+ // which §4.1's scoping of the schema to a spec P allows. But loading it
462
+ // only in paper mode meant a project shipping a schema in notes mode had
463
+ // it silently ignored: a model wrote thirty files under a namespace it
464
+ // invented while the declared fields sat at their defaults, and nothing
465
+ // said otherwise.
466
+ const resolution = specs.resolve(directory);
467
+ const spec = resolution.spec;
468
+ // Only a spec the PROJECT SHIPPED counts as a declaration. A `builtin`
469
+ // source means we fell back to a generic default, and announcing that to
470
+ // the model as "this project declares these fields" would be a false claim
471
+ // about a file the project does not have — and it would fire on every
472
+ // project without one, comparing their notes against a default's field
473
+ // names and reporting a mismatch that is an artefact of our own fallback.
474
+ const declaredFields = resolution.source === 'file' && spec !== undefined
475
+ ? Object.entries(spec.schema).map(([key, field]) => `${key} (${field.type})`)
476
+ : [];
477
+ const sink = mode !== 'paper' || spec === undefined
478
+ ? undefined
479
+ : new PaperStateSink({ store, spec, scopeFor });
480
+ // One pending correction per session. Exists in paper mode only, because in
481
+ // notes mode there is no `state_patch` for the host to reject — the model
482
+ // calls a tool instead, and the tool reports its own rejections.
483
+ //
484
+ // Keyed on the MODE, not on `spec`. Loading a spec in notes mode (so its
485
+ // declared fields can be named to the model) made `spec` defined there, and
486
+ // an empty feedback queue in notes mode is a queue nothing ever writes to
487
+ // and `take` would drain — correct by accident, and one refactor away from
488
+ // not being.
489
+ const feedback = mode !== 'paper' || spec === undefined ? undefined : new FeedbackQueue();
490
+ // Turns taken per scope without the state changing, for the drift notice.
491
+ const turnsSinceWrite = new Map();
492
+ // Applied patches per scope, so the drift diagnostic can show whether a
493
+ // notice was followed by a write — the only half of the claim that is
494
+ // actually about the model.
495
+ const stateWrites = new Map();
496
+ // §5.1's alternation. Paper mode only: in notes mode the transcript is
497
+ // intact and the model's own loop is the point, so forcing a report turn
498
+ // there would tax a mode that has no problem to solve.
499
+ const boundary = new StepBoundary();
500
+ const stepBoundaryEnabled = process.env['SKILLSTATE_STEP_BOUNDARY'] === '1';
501
+ // The step loop. Present in paper mode only, where the context is
502
+ // replaced and the model therefore cannot fall back on the transcript to
503
+ // keep going; see runtime.ts for why this belongs in code.
504
+ // `SKILLSTATE_DRIVE=0` measures the trade-off rather than assuming it:
505
+ // the paper's context replacement, with the host's own batching left
506
+ // alone. Measured 3.75x cheaper on the eight-file task with the state
507
+ // lagging the work; see CHANGELOG. Off means the step loop is not driven.
508
+ const runtime = mode === 'paper' && process.env['SKILLSTATE_DRIVE'] !== '0'
509
+ ? new RuntimeDriver({
510
+ prompt: async (sessionID, text) => {
511
+ try {
512
+ // `text` is a plain string. The type reads
513
+ // `{…}["text"]` and that indexing is the point: it IS the
514
+ // string field, not an object containing one. Passing
515
+ // `{ sessionID, text: { text } }` was rejected by the host's
516
+ // own schema with `SchemaError: Expected string at ["text"]`,
517
+ // which is why the runtime never once drove a turn — the call
518
+ // was refused every time, and the refusal was swallowed until
519
+ // a diagnostic started recording it.
520
+ await ctx.session.prompt({
521
+ sessionID,
522
+ // The text is a WAKE-UP, not the instruction. `applyPaperContext`
523
+ // clears the messages this arrives in, so the model never
524
+ // reads it — the real instruction rides in Oₜ, which is the
525
+ // paper's channel for the environment. Anything written here
526
+ // is transcript noise that a reader sees and the model does
527
+ // not, which is worse than nothing: it looks like the user
528
+ // said it.
529
+ text: '',
530
+ });
531
+ return true;
532
+ }
533
+ catch (error) {
534
+ // Swallowed for a reason — a throw here would end the event
535
+ // loop for the rest of the process — but not silently. This
536
+ // path was invisible for the whole time `session.prompt` did
537
+ // not start a turn, and an invisible failure here is
538
+ // indistinguishable from a host that simply declined.
539
+ recordPromptFailure(error);
540
+ return false;
541
+ }
542
+ },
543
+ // `SKILLSTATE_MAX_STEPS` exists because the 64-step ceiling turned
544
+ // out to be the binding constraint on a 30-file task, and a
545
+ // diagnosis that cannot be tested is a story. Measured: the model
546
+ // narrates on about 63% of steps and patches on the rest, so the
547
+ // state grows at roughly a third of the step rate — 17 files in 50
548
+ // steps, which puts 30 files at about 88 steps against a ceiling of
549
+ // 64. That is the whole of the 25/30.
550
+ maxSteps: maxStepsFromEnv(),
551
+ })
552
+ : undefined;
553
+ const maxSteps = runtime === undefined ? undefined : (maxStepsFromEnv() ?? DEFAULT_MAX_STEPS);
554
+ // Paper mode registers NO skillstate tools, and the reason is measured
555
+ // rather than doctrinal.
556
+ //
557
+ // `skillstate_update` is free-form by design — it cannot see the spec — so
558
+ // in paper mode it was a second write path into Sigma that bypasses
559
+ // `validatePatch` entirely. A thirty-file run left `total: '1523'` in the
560
+ // state file: a STRING, in a field the spec declares as `number`. No
561
+ // validated patch can produce that, so the model had used the tool, and
562
+ // nothing in the runtime noticed.
563
+ //
564
+ // §6.4's rollback guarantee is that "a rejected patch has no path into
565
+ // Sigma ... there is nothing to undo because there is nothing partially
566
+ // applied". An unvalidated second writer is exactly such a path. And the
567
+ // model does not need the tool to read: paper mode puts Sigma in the
568
+ // prompt by construction, which is the whole of eq. 1.
569
+ await ctx.tool.transform((editor) => {
570
+ // Notes mode gets the schema when the project SHIPPED one, so the tool
571
+ // that writes this file validates against the same §6.2 the paper's
572
+ // runtime uses. Only `source === 'file'` counts: a builtin spec is our own
573
+ // fallback, and holding a project's notes to it would reject notes that
574
+ // are fine. This is the same gate as `declaredFields`, for the same
575
+ // reason — a default is not a declaration.
576
+ if (mode !== 'paper') {
577
+ registerTools(editor, {
578
+ store,
579
+ sessions,
580
+ scopeFor,
581
+ // Every write resets the drift counter, in every mode. The paper-mode
582
+ // sink resets it from the event stream; this is the notes-mode path,
583
+ // and without it the notice is a false statement for the whole of
584
+ // notes mode — the state is on disk and the counter never learns it.
585
+ onWrite: (scope) => {
586
+ turnsSinceWrite.set(scope, 0);
587
+ stateWrites.set(scope, (stateWrites.get(scope) ?? 0) + 1);
233
588
  },
234
- parts: [
235
- {
236
- id: `${STATE_MESSAGE_ID}-text`,
237
- sessionID: 'skillstate',
238
- messageID: STATE_MESSAGE_ID,
239
- type: 'text',
240
- synthetic: true,
241
- text: `Current skill state (JSON): ${JSON.stringify(state)}`,
242
- },
243
- ],
244
- };
245
- // The pipeline holds the original array reference — mutate in place
246
- // (reassigning `output.messages` would not reach the LLM call).
247
- const kept = [...systemMessages, ...trimmed, stateMessage];
248
- messages.length = 0;
249
- messages.push(...kept);
250
- },
251
- // ── Compaction context injection ───────────────────────────────────
252
- // Before compaction, inject the current state into the context so the
253
- // compaction summary preserves state even after history is compressed.
254
- 'experimental.session.compacting': async (input, output) => {
255
- const agentId = pluginAgentId(input);
256
- const state = readSkillState(resolvePath(agentId));
257
- if (!Array.isArray(output.context)) {
258
- output.context = [];
589
+ ...(resolution.source === 'file' && spec !== undefined ? { schema: spec.schema } : {}),
590
+ });
591
+ }
592
+ });
593
+ // ── Session tree and the paper-mode state sink ───────────────────────
594
+ // Sub-agent sessions are created by OpenCode itself, so the parent edge
595
+ // arrives on the event stream. Until one is seen a session is treated as
596
+ // a root session, which is the correct default for single-session use.
597
+ //
598
+ // The same stream carries the completed assistant text blocks that close
599
+ // the paper's transition, so both consumers share one subscription: a
600
+ // second `subscribe()` would be a second socket for no gain.
601
+ const controller = new AbortController();
602
+ void (async () => {
603
+ try {
604
+ for await (const event of ctx.event.subscribe({ signal: controller.signal })) {
605
+ try {
606
+ recordEvent(process.env['SKILLSTATE_DEBUG_PROMPT'], event.type);
607
+ sessions.ingestEvent(event);
608
+ // A sink failure is a value, never a throw — an unhandled
609
+ // rejection here would end the loop and silently stop both the
610
+ // registry and the sink for the rest of the process's life.
611
+ //
612
+ // The outcome is NOT discarded. Every rejection reason is queued as
613
+ // corrective feedback for the next prompt, because a model whose
614
+ // patch was refused and never told about it would otherwise be
615
+ // re-shown the identical context and keep failing silently.
616
+ const outcome = sink === undefined ? undefined : await sink.ingest(event);
617
+ if (outcome !== undefined && feedback !== undefined && isTextEnded(event)) {
618
+ feedback.record(event.data.sessionID, outcome);
619
+ }
620
+ if (outcome !== undefined && isTextEnded(event)) {
621
+ const key = scopeFor(event.data.sessionID);
622
+ if (outcome.applied) {
623
+ turnsSinceWrite.set(key, 0);
624
+ stateWrites.set(key, (stateWrites.get(key) ?? 0) + 1);
625
+ boundary.patchApplied(event.data.sessionID);
626
+ // Remembered, not acted on: the turn is not over yet, and
627
+ // ordering the next step here is what made the continuation
628
+ // arrive against a request that was already superseded.
629
+ //
630
+ // `action` needs no guard — the parser refuses a response
631
+ // without one, so `applied` already implies it. The gate proved
632
+ // the guard was dead by refusing to let it be covered.
633
+ lastAction.set(event.data.sessionID, outcome.action);
634
+ }
635
+ }
636
+ // ── The runtime owns the step, and a step is not a patch ───────
637
+ //
638
+ // §5.1, lines 9–10: if no valid patch was produced, the step
639
+ // returns (Σₜ, __invalid_patch__, {invalidated: true}) — the
640
+ // state is UNCHANGED and the loop continues anyway. Advancing only
641
+ // on `applied` therefore deleted the failure case: a turn that
642
+ // produced prose instead of a patch ended the procedure, when the
643
+ // paper says it should have been retried with the reason attached.
644
+ // The feedback queue carries that reason; it just never got reached,
645
+ // because the loop stopped before the next turn.
646
+ //
647
+ // Fired on `session.step.ended`, NOT on a completed text block. A text block ends
648
+ // when the model's response ends, which is not the same thing: a
649
+ // model that narrates and then calls a tool has ended a text block
650
+ // and is nowhere near done. Advancing there ordered the next step
651
+ // while the current one was still running, and the continuation was
652
+ // consumed by a request that got superseded — measured, the
653
+ // `[next step]` marker never reached the model at all.
654
+ if (mode === 'paper' && isStepEnded(event)) {
655
+ const sessionID = event.data.sessionID;
656
+ const last = lastAction.get(sessionID);
657
+ // §5.1 lines 2–8: one step is `k + 1` attempts at the SAME Aₜ.
658
+ // A turn that produced no patch is an attempt, not a step, so the
659
+ // corrective feedback arrives on the prompt it belongs to instead
660
+ // of on the next step's entirely different one.
661
+ const verdict = runtime === undefined
662
+ ? undefined
663
+ : runtime.record(sessionID, last !== undefined);
664
+ if (verdict?.result === INVALID_PATCH) {
665
+ invalidations.set(sessionID, {
666
+ attempts: verdict.attempt,
667
+ lastError: feedback?.peek(sessionID),
668
+ });
669
+ }
670
+ if (runtime !== undefined) {
671
+ // Deferred out of the event loop: asking the server to start a
672
+ // turn from inside the handler reporting that turn is re-entrant,
673
+ // and the request is dropped.
674
+ setTimeout(() => {
675
+ void runtime
676
+ ?.advance(sessionID, last ?? CONTINUE_ACTION)
677
+ .then((step) => {
678
+ // A `null` here is the loop ending, and four different
679
+ // things end it. `max_steps` is the one that lies — a loop
680
+ // stopped at the ceiling looks exactly like a loop that ran
681
+ // its course, and that is how a 90-file run got read as a
682
+ // model that lost track of its sum at file 78 when the
683
+ // ceiling had arrived mid-file-79. Written next to the
684
+ // build stamp so a scorer never has to guess it from a
685
+ // transcript.
686
+ const stop = runtime?.lastStop;
687
+ // Both arguments are non-optional, and that is deliberate:
688
+ // this is only reached when `advance` returned null, which
689
+ // is only reached when a runtime exists, so `lastStop` is
690
+ // always set and the ceiling was resolved when the runtime
691
+ // was built. The `?.` and the `| undefined` were branches
692
+ // nothing could take.
693
+ if (step === null && stop !== undefined && maxSteps !== undefined) {
694
+ writeRunRecord(directory, stop, maxSteps);
695
+ }
696
+ });
697
+ }, 0);
698
+ }
699
+ // What the loop did, step by step. Read BEFORE the action is
700
+ // forgotten, and after the state is written, so the line answers
701
+ // the only question that matters here: did the turn before this
702
+ // one produce a patch, and what did the state say afterwards?
703
+ const snapshot = store.read(scopeFor(sessionID));
704
+ dumpStepTrace(process.env['SKILLSTATE_DEBUG_STEPS'], {
705
+ sessionID,
706
+ step: verdict?.step ?? -1,
707
+ attempt: verdict?.attempt ?? 0,
708
+ applied: last !== undefined,
709
+ done: Array.isArray(snapshot?.done) ? snapshot.done.length : -1,
710
+ total: typeof snapshot?.total === 'number' ? snapshot.total : null,
711
+ drove: runtime !== undefined,
712
+ note: last ?? CONTINUE_ACTION,
713
+ });
714
+ lastAction.delete(sessionID);
715
+ }
716
+ }
717
+ catch {
718
+ // Per EVENT, not per stream. The outer catch ends the loop, and the
719
+ // loop is the only source of the registry and the sink — so one
720
+ // malformed event used to end both for the lifetime of the process,
721
+ // silently. Found by a test that fed the loop a bare `null`.
722
+ }
259
723
  }
260
- output.context.push(`Skillstate: ${JSON.stringify(state)}`);
261
- },
262
- // ── State persistence from LLM responses ───────────────────────────
263
- // After tool execution, extract state_patch from the tool response
264
- // (output.output), and atomically merge it into the session-scoped
265
- // state (read + merge + write all inside the cross-process lock).
266
- 'tool.execute.after': async (input, output) => {
267
- const response = output.output ?? '';
268
- if (typeof response !== 'string')
269
- return;
270
- const agentId = pluginAgentId(input);
271
- const patch = extractPatch(response);
272
- if (patch) {
273
- mergeSkillState(resolvePath(agentId), patch);
724
+ }
725
+ catch {
726
+ // The stream ends when the plugin unloads or the server goes away.
727
+ // Session scoping degrades to "everyone shares the project file",
728
+ // which is safe; it must never surface as an unhandled rejection.
729
+ }
730
+ })();
731
+ // ── Context ─────────────────────────────────────────────────────────
732
+ // Registered on the agent loop only. `compaction`, `generate` and
733
+ // `title` are separate hooks in v2 and are deliberately left alone:
734
+ // after a compaction the next agent-loop request re-enters here, so
735
+ // state survives without this plugin ever touching the summariser's
736
+ // input. The same holds in paper mode, where the compaction summary is
737
+ // discarded along with the rest of the transcript.
738
+ await ctx.session.hook('context', async (event) => {
739
+ const scope = scopeFor(event.sessionID);
740
+ // Read-after-write against the host. The patch the model just emitted is
741
+ // already in this transcript, and the host does not wait for the event
742
+ // loop to deliver it, so waiting for `session.text.ended` serves the
743
+ // next request a Sigma that has not moved. Recovering it here is what
744
+ // makes the state the model is shown match the state on disk.
745
+ const recovered = await sink?.recover(event.sessionID, event.messages);
746
+ // A patch recovered here is an applied patch, so the model has accounted
747
+ // for its step and may act again. Leaving the phase alone would demand a
748
+ // second report turn for the same patch — the model would be told to
749
+ // account for something it had already reported.
750
+ if (recovered?.applied === true) {
751
+ boundary.patchApplied(event.sessionID);
752
+ }
753
+ const state = store.read(scope);
754
+ if (mode === 'paper') {
755
+ // ── §5.1's step boundary ──────────────────────────────────────────
756
+ // One request may act; the next must account for it. `tools` is
757
+ // handed to this hook on every model request, so the cycle is
758
+ // enforced by withholding the tools rather than by asking in prose.
759
+ // See step-boundary.ts for why delegation to the host's agent loop
760
+ // is not the same thing.
761
+ const target = event;
762
+ // A tool result in the incoming transcript means the host has just
763
+ // executed an action for this session. That is the observable edge of
764
+ // §5.1's `execute(aₜ, Σₜ₊₁)`, and it is what moves the session from
765
+ // `act` to `report`.
766
+ //
767
+ // OFF BY DEFAULT, and the reason is a measurement rather than a
768
+ // preference. The premise — that a model asked again with no tools
769
+ // available can only answer in text, and that text is where a patch
770
+ // lives — is false for the models tested here. Measured on the
771
+ // eight-file task with the boundary on: two text blocks, NEITHER
772
+ // containing a `state_patch`, and the model writing prose instead
773
+ // ("the saved execution state is still {total:0, files:0}… I will
774
+ // restart from src/cfg1.ts"). It had done the accounting it was asked
775
+ // for, in words, and Σ never moved. With the boundary off the same
776
+ // task reaches cfg8; with it on it stops at cfg1.
777
+ //
778
+ // So the mechanism is kept, correct and tested, behind
779
+ // SKILLSTATE_STEP_BOUNDARY=1, and the default stays the behaviour that
780
+ // measurably goes further. Turning it on is a claim to be measured, not
781
+ // a setting to leave flipped.
782
+ // Two different questions, deliberately two different answers.
783
+ //
784
+ // Did an action run? That is the boundary's question and it is about the
785
+ // turn, so it asks the turn — `hasToolResult` looks at the newest message
786
+ // and does not care whether this hook has seen it before.
787
+ //
788
+ // How many ran, for the report? That is the driver's question and it has
789
+ // to be about *new* results: this hook runs on every model request, so a
790
+ // tool message still newest across two requests would be counted twice,
791
+ // and "2 actions ran" for one `read` is exactly the kind of lie the
792
+ // report exists to stop. So the driver keeps the watermark and the
793
+ // caller does not have to know it exists.
794
+ if (hasToolResult(target.messages))
795
+ boundary.actionTaken(event.sessionID);
796
+ runtime?.noteExecutions(event.sessionID, target.messages);
797
+ if (stepBoundaryEnabled && boundary.reportRequired(event.sessionID)) {
798
+ target.tools = {};
274
799
  }
275
- },
800
+ // A session that has saved nothing yet has no Σₜ to show, and
801
+ // replacing the context with an empty state block before the agent
802
+ // has done anything would only lose the task. Stay inert.
803
+ //
804
+ // Note the feedback is deliberately NOT taken here: this early return
805
+ // happens before any prompt is built, so consuming the correction
806
+ // would drop it without ever showing it to the model.
807
+ if (Object.keys(state).length === 0)
808
+ return;
809
+ // Taken exactly once: `take` clears on read, so calling it twice would
810
+ // show the correction to nobody.
811
+ // §6.4: a step that spent all `k + 1` attempts produces a synthetic
812
+ // observation instead of the running correction, and the sentinel action
813
+ // is never executed. The invalidation is recorded by `record` on the
814
+ // turn that exhausted it, so this is where the model finally hears it.
815
+ const invalidation = invalidations.get(event.sessionID);
816
+ invalidations.delete(event.sessionID);
817
+ const correction = invalidation === undefined
818
+ ? feedback?.take(event.sessionID)
819
+ : feedback?.takeUnlessInvalidated(event.sessionID, invalidation.attempts, invalidation.lastError);
820
+ // The action the runtime is carrying out, which has to reach the model
821
+ // through Oₜ because the messages it was sent in are cleared here.
822
+ // What rides in Oₜ, and `SKILLSTATE_CONTINUATION` chooses between three
823
+ // things, because both extremes have been measured and both are wrong:
824
+ //
825
+ // unset (default) the environment's REPORT of what the runtime did
826
+ // '1' the action the model itself proposed, as an order
827
+ // '0' nothing at all
828
+ //
829
+ // §2 forbids the middle one — "the agent receives only Oₜ, never prior
830
+ // observations or actions" — and with it the model obeyed its own stored
831
+ // order: 54 reads for thirty files where a control used one grep. The
832
+ // empty end loses too: the runtime re-prompts when a turn produced a
833
+ // patch but no tool call, and with nothing saying why, the model looped
834
+ // — 98 text blocks against 43, 7.9M tokens against 1.6M. See
835
+ // `RuntimeDriver.stepReport`.
836
+ const continuationFlag = process.env['SKILLSTATE_CONTINUATION'];
837
+ const [continuationText, continuationKind] = continuationFlag === '0'
838
+ ? [undefined, 'report']
839
+ : continuationFlag === '1'
840
+ ? [runtime?.takeContinuation(event.sessionID) ?? '', 'order']
841
+ : [runtime?.stepReport(event.sessionID) ?? '', 'report'];
842
+ const raw = event.messages;
843
+ dumpPromptShape(process.env['SKILLSTATE_DEBUG_PROMPT'], raw, state);
844
+ applyPaperContext(event, buildPaperPrompt({
845
+ spec: spec,
846
+ state,
847
+ messages: raw,
848
+ // §2, on Observation: "The agent receives only Oₜ — never prior
849
+ // observations or ACTIONS."
850
+ //
851
+ // That is not an interpretation and it is not a preference. Putting
852
+ // the model's own previous action into Oₜ is putting an action into
853
+ // the channel the paper reserves for the environment's reply, and
854
+ // this implementation did it to close a real gap — the model was
855
+ // re-prompted with nothing saying why.
856
+ //
857
+ // It worked too well, which is how it was found. The model began
858
+ // answering "I'll read cfg3.ts next, as directed by the
859
+ // observation" and then reading cfg3.ts. It obeyed a stored order
860
+ // instead of choosing, made 54 `read` calls for thirty files, and
861
+ // used grep three times as a side errand. A control with no step
862
+ // driver read one file, ran ONE grep and finished in six calls. The
863
+ // order was its own past action, so it never looked for a better
864
+ // way than the one it had already written down.
865
+ ...(continuationText === undefined || continuationText.length === 0
866
+ ? {}
867
+ : { continuation: continuationText, continuationKind }),
868
+ ...(correction === undefined ? {} : { feedback: correction }),
869
+ }), HOST_ACTION_NOTE);
870
+ return;
871
+ }
872
+ if (!store.exists(scope))
873
+ return;
874
+ // ── Drift detection ───────────────────────────────────────────────
875
+ // A user who initialized skillstate did so because the work needs
876
+ // cross-turn memory. An agent that then quietly stops writing drifts
877
+ // back to a growing transcript and pays for it in re-sent tokens. The
878
+ // fix is to notice and say so — a measured fact, not an instruction,
879
+ // so it cannot displace the task the way the v1 injection did.
880
+ //
881
+ // Counted per scope and reset by the sink on every applied patch, so
882
+ // a sub-agent's writes do not silence the main session's counter.
883
+ const sinceWrite = (turnsSinceWrite.get(scope) ?? 0) + 1;
884
+ turnsSinceWrite.set(scope, sinceWrite);
885
+ const hint = buildStateHint({
886
+ state,
887
+ statePath: path.relative(store.projectDirectory, store.pathFor(scope)),
888
+ scope,
889
+ declaredFields,
890
+ // A state file on disk is the definition of an initialized project:
891
+ // the user ran `skillstate init`, or something wrote one.
892
+ initialized: store.exists(scope),
893
+ turnsSinceWrite: sinceWrite,
894
+ });
895
+ if (hint.length === 0)
896
+ return;
897
+ event.system.push({ type: 'text', text: hint });
898
+ // What went out, and what the model had done about it at the time. The
899
+ // only way to tell "the notice was ignored" from "the notice was never
900
+ // built" — the two are indistinguishable from the outside otherwise.
901
+ dumpDrift(process.env['SKILLSTATE_DEBUG_DRIFT'], {
902
+ scope,
903
+ turns: sinceWrite,
904
+ notice: hint.includes(driftNotice(sinceWrite)),
905
+ writes: stateWrites.get(scope) ?? 0,
906
+ });
907
+ });
908
+ return () => {
909
+ controller.abort();
276
910
  };
277
- };
278
- }
911
+ },
912
+ });
913
+ export default SkillStatePlugin;
279
914
  //# sourceMappingURL=plugin.js.map