@skillstate/opencode 3.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +196 -131
- package/dist/feedback.d.ts +178 -0
- package/dist/feedback.d.ts.map +1 -0
- package/dist/feedback.js +235 -0
- package/dist/feedback.js.map +1 -0
- package/dist/index.d.ts +33 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +27 -3
- package/dist/index.js.map +1 -1
- package/dist/mode.d.ts +97 -0
- package/dist/mode.d.ts.map +1 -0
- package/dist/mode.js +111 -0
- package/dist/mode.js.map +1 -0
- package/dist/opencode-adapter.d.ts +4 -3
- package/dist/opencode-adapter.d.ts.map +1 -1
- package/dist/opencode-adapter.js.map +1 -1
- package/dist/paper-mode.d.ts +421 -0
- package/dist/paper-mode.d.ts.map +1 -0
- package/dist/paper-mode.js +445 -0
- package/dist/paper-mode.js.map +1 -0
- package/dist/plugin.d.ts +158 -8
- package/dist/plugin.d.ts.map +1 -1
- package/dist/plugin.js +788 -21
- package/dist/plugin.js.map +1 -1
- package/dist/response-sink.d.ts +208 -0
- package/dist/response-sink.d.ts.map +1 -0
- package/dist/response-sink.js +243 -0
- package/dist/response-sink.js.map +1 -0
- package/dist/runtime.d.ts +203 -0
- package/dist/runtime.d.ts.map +1 -0
- package/dist/runtime.js +332 -0
- package/dist/runtime.js.map +1 -0
- package/dist/spec-loader.d.ts +81 -0
- package/dist/spec-loader.d.ts.map +1 -0
- package/dist/spec-loader.js +163 -0
- package/dist/spec-loader.js.map +1 -0
- package/dist/step-boundary.d.ts +91 -0
- package/dist/step-boundary.d.ts.map +1 -0
- package/dist/step-boundary.js +109 -0
- package/dist/step-boundary.js.map +1 -0
- package/dist/system-hint.d.ts +70 -3
- package/dist/system-hint.d.ts.map +1 -1
- package/dist/system-hint.js +90 -11
- package/dist/system-hint.js.map +1 -1
- package/dist/tools.d.ts +35 -1
- package/dist/tools.d.ts.map +1 -1
- package/dist/tools.js +16 -0
- package/dist/tools.js.map +1 -1
- package/package.json +1 -1
package/dist/plugin.js
CHANGED
|
@@ -27,14 +27,33 @@
|
|
|
27
27
|
*
|
|
28
28
|
* ── The v2 design ────────────────────────────────────────────────────────
|
|
29
29
|
*
|
|
30
|
+
* Two modes, each enforced by a test. They are different contracts with the
|
|
31
|
+
* model, not variants of one behaviour:
|
|
32
|
+
*
|
|
33
|
+
* - **`notes` (default).** Contribute one additive, bounded fragment to
|
|
34
|
+
* `event.system` and leave the transcript alone. The agent sees its own
|
|
35
|
+
* history the way the host intends, and the saved notes ride alongside it.
|
|
36
|
+
* This is the mode that fixed the v1 failure, and it is the default for
|
|
37
|
+
* exactly that reason.
|
|
38
|
+
* - **`paper` (opt-in).** Replace the model-facing context with
|
|
39
|
+
* Aₜ = (P, Σₜ, Oₜ) — the paper's Appendix A.4 prompt, byte-verbatim — and
|
|
40
|
+
* apply the `state_patch` the model emits in response. This is the paper's
|
|
41
|
+
* specification, and it is a real behavioural change: the model stops
|
|
42
|
+
* seeing its transcript, because §3.2 discards the reasoning trace by
|
|
43
|
+
* construction. Select it with `mode: "paper"` in the project's
|
|
44
|
+
* `skillstate.json` or `SKILLSTATE_MODE=paper`; see `mode.ts`.
|
|
45
|
+
*
|
|
46
|
+
* The default is `notes` and must stay that way: a default that discards the
|
|
47
|
+
* user's task is the v1 bug under a new name.
|
|
48
|
+
*
|
|
30
49
|
* Three rules, each enforced by a test:
|
|
31
50
|
*
|
|
32
|
-
* - **
|
|
33
|
-
* fragment to `event.system` and leaves the transcript alone.
|
|
34
|
-
* `tests/opencode/context-integrity.test.ts`.
|
|
35
|
-
* - **Never inject behavioural instructions.** The system
|
|
36
|
-
* describes what the notes are and when to use them; it contains
|
|
37
|
-
* "you must", no "always", and no output format. See
|
|
51
|
+
* - **Notes mode never mutates `event.messages`.** The plugin contributes
|
|
52
|
+
* one additive fragment to `event.system` and leaves the transcript alone.
|
|
53
|
+
* See `tests/opencode/context-integrity.test.ts`.
|
|
54
|
+
* - **Never inject behavioural instructions in notes mode.** The system
|
|
55
|
+
* fragment describes what the notes are and when to use them; it contains
|
|
56
|
+
* no "you must", no "always", and no output format. See
|
|
38
57
|
* `system-hint.ts`.
|
|
39
58
|
* - **Inert until used.** A project with no state file gets no system
|
|
40
59
|
* fragment at all and behaves exactly like vanilla OpenCode. No files are
|
|
@@ -64,24 +83,348 @@
|
|
|
64
83
|
* ```
|
|
65
84
|
*/
|
|
66
85
|
import { Plugin } from '@opencode/plugin';
|
|
86
|
+
import * as fs from 'node:fs';
|
|
67
87
|
import * as path from 'node:path';
|
|
88
|
+
import { fileURLToPath } from 'node:url';
|
|
89
|
+
import { resolvePluginMode } from './mode.js';
|
|
90
|
+
import { HOST_ACTION_NOTE, applyPaperContext, buildPaperPrompt, latestObservation } from './paper-mode.js';
|
|
91
|
+
import { FeedbackQueue } from './feedback.js';
|
|
92
|
+
import { PaperStateSink, isTextEnded } from './response-sink.js';
|
|
68
93
|
import { SessionRegistry, stateScopeFor } from './session-registry.js';
|
|
94
|
+
import { SpecResolver } from './spec-loader.js';
|
|
95
|
+
import { DEFAULT_MAX_STEPS, INVALID_PATCH, RuntimeDriver } from './runtime.js';
|
|
96
|
+
import { StepBoundary } from './step-boundary.js';
|
|
69
97
|
import { ProjectStateStore } from './state-store.js';
|
|
70
|
-
import { buildStateHint } from './system-hint.js';
|
|
98
|
+
import { buildStateHint, driftNotice } from './system-hint.js';
|
|
71
99
|
import { registerTools } from './tools.js';
|
|
72
100
|
/** Stable plugin id — scopes plugin storage and identifies it in `/api/plugin`. */
|
|
73
101
|
export const PLUGIN_ID = 'skillstate';
|
|
102
|
+
/**
|
|
103
|
+
* The action carried forward when a turn produced no usable patch.
|
|
104
|
+
*
|
|
105
|
+
* NOT `__invalid_patch__`, though the paper names that sentinel at §5.1 line
|
|
106
|
+
* 9. It is a return value there — what the step function hands back to signal
|
|
107
|
+
* that Σ is unchanged — and forwarding it into the prompt as the next action
|
|
108
|
+
* is meaningless to a model: it is a name, not a request. The retry instruction
|
|
109
|
+
* the model actually needs already rides in Oₜ through the feedback queue, so
|
|
110
|
+
* this only has to say "keep going", and the queue says why.
|
|
111
|
+
*/
|
|
112
|
+
export const CONTINUE_ACTION = 'continue';
|
|
113
|
+
/**
|
|
114
|
+
* The action the model last asked for, per session, waiting for the turn to end.
|
|
115
|
+
*
|
|
116
|
+
* A text block ending is not a turn ending — a model that narrates and then
|
|
117
|
+
* calls a tool ends a block and is nowhere near done. The host says when the
|
|
118
|
+
* turn is actually over, and that is the only point at which asking for the
|
|
119
|
+
* next step is correct.
|
|
120
|
+
*/
|
|
121
|
+
const lastAction = new Map();
|
|
122
|
+
/**
|
|
123
|
+
* Sessions whose step spent all `k + 1` attempts, for §6.4's synthetic
|
|
124
|
+
* observation. Session-scoped rather than global because the attempt budget is
|
|
125
|
+
* per session: one session stalling must not put an invalidation in front of
|
|
126
|
+
* another's next prompt.
|
|
127
|
+
*/
|
|
128
|
+
const invalidations = new Map();
|
|
129
|
+
/**
|
|
130
|
+
* Whether an event says the host has finished a step.
|
|
131
|
+
*
|
|
132
|
+
* `session.step.ended`, measured — not `session.idle`, which is what the SDK
|
|
133
|
+
* type reads like and which the host never emits. Recorded every event type the
|
|
134
|
+
* plugin receives for one run: 2x `session.step.ended`, 2x `session.text.ended`,
|
|
135
|
+
* and zero of `session.idle`. So the trigger that was supposed to turn the loop
|
|
136
|
+
* never fired once, and the loop could not turn. The SDK exports
|
|
137
|
+
* `SessionMessageIdle`, which is a different thing entirely and reads like an
|
|
138
|
+
* event name because it is not one.
|
|
139
|
+
*/
|
|
140
|
+
/**
|
|
141
|
+
* Write `.skillstate/.build.json`: the plugin id, the package version, and the
|
|
142
|
+
* build's own mtime and size.
|
|
143
|
+
*
|
|
144
|
+
* Deliberately cheap and deliberately once. The mtime is the whole point — it is
|
|
145
|
+
* the thing that differs between two builds of identical source, which is exactly
|
|
146
|
+
* the case a version string cannot see.
|
|
147
|
+
*
|
|
148
|
+
* Never throws and never blocks setup: a stamp that fails to write is a missing
|
|
149
|
+
* field in a diagnostic file, and a plugin that cannot start is worse.
|
|
150
|
+
*/
|
|
151
|
+
function writeBuildStamp(directory) {
|
|
152
|
+
try {
|
|
153
|
+
// The module's OWN file, not a guessed sibling. The first version of this
|
|
154
|
+
// stat'd `path.join(here, 'index.js')` — right under `dist/`, absent under
|
|
155
|
+
// `src/` — so the try/catch swallowed ENOENT and the function was silently
|
|
156
|
+
// dead in every test that ran it. The same shape as this whole day: guessing
|
|
157
|
+
// at a path instead of measuring the thing already in hand.
|
|
158
|
+
const entry = fileURLToPath(import.meta.url);
|
|
159
|
+
const here = path.dirname(entry);
|
|
160
|
+
const stat = fs.statSync(entry);
|
|
161
|
+
// No inner try: the outer one already covers this, and a stamp that cannot be
|
|
162
|
+
// written is a missing diagnostic, not a failure. The inner catch returned
|
|
163
|
+
// 'unknown', which is a number no run could ever check against anything — an
|
|
164
|
+
// unreachvable branch that the coverage gate was right to refuse.
|
|
165
|
+
// `version` is optional in a package.json, so it may be absent, and absence
|
|
166
|
+
// is the honest rendering: JSON.stringify drops an undefined field on its
|
|
167
|
+
// own. `?? 'unknown'` produced a string no run could check against anything,
|
|
168
|
+
// and a ternary guarding the field produced a second unreachable branch for
|
|
169
|
+
// the coverage gate to refuse. Both are the same mistake — a fallback value
|
|
170
|
+
// where a missing one was already correct.
|
|
171
|
+
const version = JSON.parse(fs.readFileSync(path.join(here, '..', 'package.json'), 'utf-8')).version;
|
|
172
|
+
// Only into an EXISTING `.skillstate/`. The plugin is inert for a project
|
|
173
|
+
// that never ran `skillstate init` — and it uses the presence of that
|
|
174
|
+
// directory as the definition of "initialised". Creating it here would
|
|
175
|
+
// initialise every project the plugin is installed into, which is not a
|
|
176
|
+
// diagnostic, it is a side effect with consequences.
|
|
177
|
+
//
|
|
178
|
+
// Caught by an existing test, "creates no files in a project that has no
|
|
179
|
+
// state", three lines after this was written. A run with a state file is a
|
|
180
|
+
// run worth stamping; a project that has none is not a run at all.
|
|
181
|
+
const dir = path.join(directory, '.skillstate');
|
|
182
|
+
if (!fs.existsSync(dir))
|
|
183
|
+
return;
|
|
184
|
+
fs.writeFileSync(path.join(dir, '.build.json'), `${JSON.stringify({ plugin: PLUGIN_ID, version, distMtimeMs: stat.mtimeMs, distBytes: stat.size }, null, 2)}\n`);
|
|
185
|
+
}
|
|
186
|
+
catch {
|
|
187
|
+
// A diagnostic that cannot be written is not a reason to refuse to run.
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* Write `.skillstate/.run.json`: why the paper-mode loop last declined to spend
|
|
192
|
+
* a step, and the ceiling it was running under.
|
|
193
|
+
*
|
|
194
|
+
* The loop's end is otherwise invisible. `advance` returns `null` for a terminal
|
|
195
|
+
* action, a host refusal, a turn with no action, and the ceiling, and the
|
|
196
|
+
* caller cannot tell them from the return value — so a run stopped at the
|
|
197
|
+
* ceiling leaves a transcript that looks exactly like a run that finished. That
|
|
198
|
+
* is measured, not hypothetical: a 90-file run stopped at step 100 mid-file-79
|
|
199
|
+
* and was read as a model that lost track of its running sum at file 78.
|
|
200
|
+
*
|
|
201
|
+
* Same rules as the build stamp: only into an existing `.skillstate/`, and never
|
|
202
|
+
* allowed to throw. A run record that cannot be written is a missing diagnostic,
|
|
203
|
+
* not a reason to take the session down.
|
|
204
|
+
*/
|
|
205
|
+
function writeRunRecord(directory, stop, maxSteps) {
|
|
206
|
+
try {
|
|
207
|
+
const dir = path.join(directory, '.skillstate');
|
|
208
|
+
if (!fs.existsSync(dir))
|
|
209
|
+
return;
|
|
210
|
+
fs.writeFileSync(path.join(dir, '.run.json'), `${JSON.stringify({ stop, maxSteps }, null, 2)}\n`);
|
|
211
|
+
}
|
|
212
|
+
catch {
|
|
213
|
+
// A diagnostic that cannot be written is not a reason to refuse to run. A
|
|
214
|
+
// `.skillstate` that is a FILE rather than a directory passes the
|
|
215
|
+
// `existsSync` above and fails the write here, and a run must survive that.
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
function isStepEnded(event) {
|
|
219
|
+
if (typeof event !== 'object' || event === null)
|
|
220
|
+
return false;
|
|
221
|
+
const typed = event;
|
|
222
|
+
return (typed.type === 'session.step.ended' &&
|
|
223
|
+
typeof typed.data?.sessionID === 'string');
|
|
224
|
+
}
|
|
225
|
+
/**
|
|
226
|
+
* Record a failed step request, so "the host declined" is not a guess.
|
|
227
|
+
*
|
|
228
|
+
* @non-paper diagnostics. Enabled by `SKILLSTATE_DEBUG_PROMPT`, appended to
|
|
229
|
+
* the same file, tagged so it cannot be mistaken for a prompt record.
|
|
230
|
+
*/
|
|
231
|
+
function recordPromptFailure(error) {
|
|
232
|
+
const path = process.env['SKILLSTATE_DEBUG_PROMPT'];
|
|
233
|
+
if (path === undefined || path.length === 0)
|
|
234
|
+
return;
|
|
235
|
+
const message = error instanceof Error ? `${error.name}: ${error.message}` : String(error);
|
|
236
|
+
try {
|
|
237
|
+
fs.appendFileSync(path, `${JSON.stringify({ promptFailure: message })}\n`);
|
|
238
|
+
}
|
|
239
|
+
catch {
|
|
240
|
+
// Diagnostics must never break the agent loop.
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
/**
|
|
244
|
+
* Record every event type the plugin actually receives.
|
|
245
|
+
*
|
|
246
|
+
* @non-paper diagnostics, same file. The advance is triggered by one event
|
|
247
|
+
* type and one, and a trigger that never fires is indistinguishable from one
|
|
248
|
+
* that is wired wrong — so the arrival counts have to be visible. Cheap, and
|
|
249
|
+
* it would have saved guessing.
|
|
250
|
+
*/
|
|
251
|
+
export function recordEvent(path, type) {
|
|
252
|
+
if (path === undefined || path.length === 0)
|
|
253
|
+
return;
|
|
254
|
+
try {
|
|
255
|
+
fs.appendFileSync(path, `${JSON.stringify({ event: type })}\n`);
|
|
256
|
+
}
|
|
257
|
+
catch {
|
|
258
|
+
// Diagnostics must never break the agent loop.
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* Whether the host has just executed a tool for this request.
|
|
263
|
+
*
|
|
264
|
+
* A tool result in the transcript is the observable edge of "an action ran".
|
|
265
|
+
* There is no event that says so in a shape this plugin can trust — and an
|
|
266
|
+
* event the host does not wait for is what caused the read-after-write race
|
|
267
|
+
* fixed in `response-sink.ts`, so the transcript is the more reliable of the
|
|
268
|
+
* two here as well as the more available one.
|
|
269
|
+
*/
|
|
270
|
+
/** Tool-result parts in the newest tool message, counted. */
|
|
271
|
+
export function countToolResults(messages) {
|
|
272
|
+
const last = messages[messages.length - 1];
|
|
273
|
+
if (last === undefined || last.role !== 'tool')
|
|
274
|
+
return 0;
|
|
275
|
+
if (!Array.isArray(last.content))
|
|
276
|
+
return 0;
|
|
277
|
+
return last.content.filter((part) => typeof part === 'object' &&
|
|
278
|
+
part !== null &&
|
|
279
|
+
String(part.type).startsWith('tool-result')).length;
|
|
280
|
+
}
|
|
281
|
+
/** Whether the newest message carries a tool result. One definition, one use. */
|
|
282
|
+
function hasToolResult(messages) {
|
|
283
|
+
return countToolResults(messages) > 0;
|
|
284
|
+
}
|
|
285
|
+
/** How much of an observation the diagnostic records. */
|
|
286
|
+
const DEBUG_OBSERVATION_CHARS = 400;
|
|
287
|
+
/**
|
|
288
|
+
* Append what the host actually handed us to a file, for diagnosis.
|
|
289
|
+
*
|
|
290
|
+
* @non-paper diagnostics. Enabled by `SKILLSTATE_DEBUG_PROMPT=<path>`.
|
|
291
|
+
*
|
|
292
|
+
* This exists because of a bug that was invisible from the inside for a
|
|
293
|
+
* long time. The model would run a tool, get the answer, and never record
|
|
294
|
+
* it — which looks exactly like a model refusing to cooperate, and sent the
|
|
295
|
+
* search through prompt slots, model choice and spec wording. The cause was
|
|
296
|
+
* the SHAPE: OpenCode v2 delivers a tool result as
|
|
297
|
+
* `{ type: 'tool-result', result: { value } }`, so a reader that only knew
|
|
298
|
+
* `{ type: 'text', text }` made Oₜ permanently empty without ever throwing.
|
|
299
|
+
*
|
|
300
|
+
* The dump records the part types alongside the extracted text, so that
|
|
301
|
+
* class of failure is visible on sight: a `tool-result` in the list next to
|
|
302
|
+
* an empty `observation` says the reader, not the model, is at fault.
|
|
303
|
+
*
|
|
304
|
+
* Append-only so a session's turns accumulate in order, and every failure
|
|
305
|
+
* is swallowed — diagnostics must never break the agent loop.
|
|
306
|
+
*/
|
|
307
|
+
export function dumpPromptShape(path, messages, state) {
|
|
308
|
+
if (path === undefined || path.length === 0)
|
|
309
|
+
return;
|
|
310
|
+
const record = {
|
|
311
|
+
turn: messages.length,
|
|
312
|
+
// Σ as the model was shown it, not as it ended up on disk. A model that
|
|
313
|
+
// writes back a stale total is indistinguishable from one that was never
|
|
314
|
+
// given a fresh one, and those are opposite bugs.
|
|
315
|
+
state,
|
|
316
|
+
roles: messages.map((m) => m.role),
|
|
317
|
+
partTypes: messages.map((m) => Array.isArray(m.content)
|
|
318
|
+
? m.content.map((part) => typeof part === 'object' && part !== null
|
|
319
|
+
? String(part.type)
|
|
320
|
+
: typeof part)
|
|
321
|
+
: typeof m.content),
|
|
322
|
+
observation: latestObservation(messages).content.slice(0, DEBUG_OBSERVATION_CHARS),
|
|
323
|
+
};
|
|
324
|
+
try {
|
|
325
|
+
fs.appendFileSync(path, `${JSON.stringify(record)}\n`);
|
|
326
|
+
}
|
|
327
|
+
catch {
|
|
328
|
+
// Diagnostics must never break the agent loop.
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
/**
|
|
332
|
+
* One line per turn of the anti-drift diagnostic.
|
|
333
|
+
*
|
|
334
|
+
* The drift notice has a claim attached to it — "the model drifts, the notice
|
|
335
|
+
* brings it back" — and neither half can be checked from inside the process.
|
|
336
|
+
* A notice in a prompt is not an observation of a model: the fragment may be
|
|
337
|
+
* built correctly and the model may ignore it, and the two look identical
|
|
338
|
+
* from the code's side. Worse, both look identical from the *outside* too,
|
|
339
|
+
* which is what made the earlier `tool-result` bug so expensive to find.
|
|
340
|
+
*
|
|
341
|
+
* So each line carries the evidence that distinguishes them:
|
|
342
|
+
*
|
|
343
|
+
* - `notice` — was the drift sentence in the fragment that went out this turn;
|
|
344
|
+
* - `writes` — how many times the state file had changed when it went out, so
|
|
345
|
+
* a notice that repeats forever is visible as a flat counter;
|
|
346
|
+
* - `fragments` — how many turns had passed without a change.
|
|
347
|
+
*
|
|
348
|
+
* That is enough to say "the notice fired and the state moved afterwards" or
|
|
349
|
+
* "the notice fired and nothing happened", which is the only claim worth
|
|
350
|
+
* making about it.
|
|
351
|
+
*
|
|
352
|
+
* @non-paper diagnostics. Enabled by `SKILLSTATE_DEBUG_DRIFT=<path>`.
|
|
353
|
+
* Separate from {@link dumpPromptShape} because it answers a different
|
|
354
|
+
* question: that one asks what the host sent, this one asks what the model
|
|
355
|
+
* did about what we sent.
|
|
356
|
+
*/
|
|
357
|
+
export function dumpDrift(path, record) {
|
|
358
|
+
if (path === undefined || path.length === 0)
|
|
359
|
+
return;
|
|
360
|
+
try {
|
|
361
|
+
fs.appendFileSync(path, `${JSON.stringify(record)}\n`);
|
|
362
|
+
}
|
|
363
|
+
catch {
|
|
364
|
+
// Diagnostics must never break the agent loop.
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
/**
|
|
368
|
+
* One line per step, for the run that answered correctly while its state
|
|
369
|
+
* under-reported the work. Enabled by `SKILLSTATE_DEBUG_STEPS=<path>`.
|
|
370
|
+
*
|
|
371
|
+
* The 30-file measurement produced the most confusing result in this project's
|
|
372
|
+
* history: paper mode answered correctly, its state ended 25/30, and the
|
|
373
|
+
* control's ended 30/30 complete. Twenty-five patches were emitted and all
|
|
374
|
+
* twenty-five landed, so no patch was lost — the model read every file and
|
|
375
|
+
* declined to patch the last five, while the loop kept driving it. From the
|
|
376
|
+
* outside that is indistinguishable from the loop stopping, from the model
|
|
377
|
+
* silently abandoning the protocol, and from the ceiling being hit.
|
|
378
|
+
*
|
|
379
|
+
* So this prints the thing that tells those apart: at every step, whether a
|
|
380
|
+
* patch was applied, what the state looked like, and whether the driver asked
|
|
381
|
+
* again. Every previous wrong guess in this file came from reasoning about the
|
|
382
|
+
* loop instead of watching it.
|
|
383
|
+
*/
|
|
384
|
+
export function dumpStepTrace(path, record) {
|
|
385
|
+
if (path === undefined || path.length === 0)
|
|
386
|
+
return;
|
|
387
|
+
try {
|
|
388
|
+
fs.appendFileSync(path, `${JSON.stringify(record)}\n`);
|
|
389
|
+
}
|
|
390
|
+
catch {
|
|
391
|
+
// Diagnostics must never break the agent loop.
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
/**
|
|
395
|
+
* The step ceiling, or `undefined` to keep the default.
|
|
396
|
+
*
|
|
397
|
+
* A malformed value is ignored rather than thrown on or silently clamped: a
|
|
398
|
+
* typo in an environment variable should leave the ceiling where the code says
|
|
399
|
+
* it is, not quietly become some other number that then gets measured.
|
|
400
|
+
*/
|
|
401
|
+
export function maxStepsFromEnv() {
|
|
402
|
+
const raw = process.env['SKILLSTATE_MAX_STEPS'];
|
|
403
|
+
if (raw === undefined || raw.length === 0)
|
|
404
|
+
return undefined;
|
|
405
|
+
// `Number`, not `parseInt`: parseInt('12.5') is 12, which is the silent
|
|
406
|
+
// truncation the comment above warns against — a ceiling set to a fifth more
|
|
407
|
+
// than asked for, measured and reported as if it were what was asked.
|
|
408
|
+
const parsed = Number(raw);
|
|
409
|
+
return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : undefined;
|
|
410
|
+
}
|
|
74
411
|
/**
|
|
75
412
|
* The plugin definition.
|
|
76
413
|
*
|
|
77
|
-
* `setup` wires
|
|
414
|
+
* `setup` wires the session registry, the project state store, the native
|
|
415
|
+
* tools, the mode resolver and the one `context` hook, then returns a cleanup
|
|
416
|
+
* function.
|
|
78
417
|
*
|
|
79
418
|
* - a {@link SessionRegistry}, fed by the server event stream, so a
|
|
80
419
|
* sub-agent session is recognised and given its own state file;
|
|
81
420
|
* - a {@link ProjectStateStore} rooted at the plugin's own project
|
|
82
421
|
* location, so two checkouts served by one OpenCode server never share
|
|
83
422
|
* state;
|
|
84
|
-
* -
|
|
423
|
+
* - a {@link SpecResolver} for paper mode's P, so a project that ships its
|
|
424
|
+
* own `skill-spec.json` gets its own procedure;
|
|
425
|
+
* - a {@link PaperStateSink}, in paper mode only, which applies the
|
|
426
|
+
* `state_patch` the model emits;
|
|
427
|
+
* - native tools plus a single `context` hook whose body depends on the mode.
|
|
85
428
|
*
|
|
86
429
|
* The event subscription is the only resource the plugin owns, so the
|
|
87
430
|
* returned cleanup aborts it. Hook and tool registrations are disposed by
|
|
@@ -95,21 +438,288 @@ export const SkillStatePlugin = Plugin.define({
|
|
|
95
438
|
// `ctx.location.project.canonical` is the canonical checkout, stable
|
|
96
439
|
// across worktrees and symlinks. The v1 plugin used `process.cwd()`,
|
|
97
440
|
// which in v2 is the server's cwd, not the session's project.
|
|
98
|
-
const
|
|
99
|
-
|
|
100
|
-
|
|
441
|
+
const directory = ctx.location.project.canonical;
|
|
442
|
+
const store = new ProjectStateStore({ directory });
|
|
443
|
+
// Stamp the build, once, so a run records WHICH plugin it used.
|
|
444
|
+
//
|
|
445
|
+
// The host resolves plugins by workspace rather than by the name in
|
|
446
|
+
// `opencode.json` (measured: asking for a package that does not exist loads
|
|
447
|
+
// the one that does), and the plugin loads from `dist/`. So a run's
|
|
448
|
+
// behaviour depends on a build that is not named anywhere in its own output,
|
|
449
|
+
// and a fix that was committed without a rebuild produces a measurement of
|
|
450
|
+
// the previous version while looking like a measurement of this one.
|
|
451
|
+
//
|
|
452
|
+
// This project has already paid for that twice: an alias gap made the bench
|
|
453
|
+
// tests silently load a stale dist, and a fixture knob was measured after
|
|
454
|
+
// the stand had been seeded with an older fixture. Both were invisible.
|
|
455
|
+
void writeBuildStamp(directory);
|
|
456
|
+
const mode = resolvePluginMode({ directory }).mode;
|
|
457
|
+
const specs = new SpecResolver();
|
|
458
|
+
// Resolved in BOTH modes now, and the difference is what each one does
|
|
459
|
+
// with it. Paper mode formats P into the prompt and validates every patch
|
|
460
|
+
// against the schema; notes mode has no P to format and does not enforce,
|
|
461
|
+
// which §4.1's scoping of the schema to a spec P allows. But loading it
|
|
462
|
+
// only in paper mode meant a project shipping a schema in notes mode had
|
|
463
|
+
// it silently ignored: a model wrote thirty files under a namespace it
|
|
464
|
+
// invented while the declared fields sat at their defaults, and nothing
|
|
465
|
+
// said otherwise.
|
|
466
|
+
const resolution = specs.resolve(directory);
|
|
467
|
+
const spec = resolution.spec;
|
|
468
|
+
// Only a spec the PROJECT SHIPPED counts as a declaration. A `builtin`
|
|
469
|
+
// source means we fell back to a generic default, and announcing that to
|
|
470
|
+
// the model as "this project declares these fields" would be a false claim
|
|
471
|
+
// about a file the project does not have — and it would fire on every
|
|
472
|
+
// project without one, comparing their notes against a default's field
|
|
473
|
+
// names and reporting a mismatch that is an artefact of our own fallback.
|
|
474
|
+
const declaredFields = resolution.source === 'file' && spec !== undefined
|
|
475
|
+
? Object.entries(spec.schema).map(([key, field]) => `${key} (${field.type})`)
|
|
476
|
+
: [];
|
|
477
|
+
const sink = mode !== 'paper' || spec === undefined
|
|
478
|
+
? undefined
|
|
479
|
+
: new PaperStateSink({ store, spec, scopeFor });
|
|
480
|
+
// One pending correction per session. Exists in paper mode only, because in
|
|
481
|
+
// notes mode there is no `state_patch` for the host to reject — the model
|
|
482
|
+
// calls a tool instead, and the tool reports its own rejections.
|
|
483
|
+
//
|
|
484
|
+
// Keyed on the MODE, not on `spec`. Loading a spec in notes mode (so its
|
|
485
|
+
// declared fields can be named to the model) made `spec` defined there, and
|
|
486
|
+
// an empty feedback queue in notes mode is a queue nothing ever writes to
|
|
487
|
+
// and `take` would drain — correct by accident, and one refactor away from
|
|
488
|
+
// not being.
|
|
489
|
+
const feedback = mode !== 'paper' || spec === undefined ? undefined : new FeedbackQueue();
|
|
490
|
+
// Turns taken per scope without the state changing, for the drift notice.
|
|
491
|
+
const turnsSinceWrite = new Map();
|
|
492
|
+
// Applied patches per scope, so the drift diagnostic can show whether a
|
|
493
|
+
// notice was followed by a write — the only half of the claim that is
|
|
494
|
+
// actually about the model.
|
|
495
|
+
const stateWrites = new Map();
|
|
496
|
+
// §5.1's alternation. Paper mode only: in notes mode the transcript is
|
|
497
|
+
// intact and the model's own loop is the point, so forcing a report turn
|
|
498
|
+
// there would tax a mode that has no problem to solve.
|
|
499
|
+
const boundary = new StepBoundary();
|
|
500
|
+
const stepBoundaryEnabled = process.env['SKILLSTATE_STEP_BOUNDARY'] === '1';
|
|
501
|
+
// The step loop. Present in paper mode only, where the context is
|
|
502
|
+
// replaced and the model therefore cannot fall back on the transcript to
|
|
503
|
+
// keep going; see runtime.ts for why this belongs in code.
|
|
504
|
+
// `SKILLSTATE_DRIVE=0` measures the trade-off rather than assuming it:
|
|
505
|
+
// the paper's context replacement, with the host's own batching left
|
|
506
|
+
// alone. Measured 3.75x cheaper on the eight-file task with the state
|
|
507
|
+
// lagging the work; see CHANGELOG. Off means the step loop is not driven.
|
|
508
|
+
const runtime = mode === 'paper' && process.env['SKILLSTATE_DRIVE'] !== '0'
|
|
509
|
+
? new RuntimeDriver({
|
|
510
|
+
prompt: async (sessionID, text) => {
|
|
511
|
+
try {
|
|
512
|
+
// `text` is a plain string. The type reads
|
|
513
|
+
// `{…}["text"]` and that indexing is the point: it IS the
|
|
514
|
+
// string field, not an object containing one. Passing
|
|
515
|
+
// `{ sessionID, text: { text } }` was rejected by the host's
|
|
516
|
+
// own schema with `SchemaError: Expected string at ["text"]`,
|
|
517
|
+
// which is why the runtime never once drove a turn — the call
|
|
518
|
+
// was refused every time, and the refusal was swallowed until
|
|
519
|
+
// a diagnostic started recording it.
|
|
520
|
+
await ctx.session.prompt({
|
|
521
|
+
sessionID,
|
|
522
|
+
// The text is a WAKE-UP, not the instruction. `applyPaperContext`
|
|
523
|
+
// clears the messages this arrives in, so the model never
|
|
524
|
+
// reads it — the real instruction rides in Oₜ, which is the
|
|
525
|
+
// paper's channel for the environment. Anything written here
|
|
526
|
+
// is transcript noise that a reader sees and the model does
|
|
527
|
+
// not, which is worse than nothing: it looks like the user
|
|
528
|
+
// said it.
|
|
529
|
+
text: '',
|
|
530
|
+
});
|
|
531
|
+
return true;
|
|
532
|
+
}
|
|
533
|
+
catch (error) {
|
|
534
|
+
// Swallowed for a reason — a throw here would end the event
|
|
535
|
+
// loop for the rest of the process — but not silently. This
|
|
536
|
+
// path was invisible for the whole time `session.prompt` did
|
|
537
|
+
// not start a turn, and an invisible failure here is
|
|
538
|
+
// indistinguishable from a host that simply declined.
|
|
539
|
+
recordPromptFailure(error);
|
|
540
|
+
return false;
|
|
541
|
+
}
|
|
542
|
+
},
|
|
543
|
+
// `SKILLSTATE_MAX_STEPS` exists because the 64-step ceiling turned
|
|
544
|
+
// out to be the binding constraint on a 30-file task, and a
|
|
545
|
+
// diagnosis that cannot be tested is a story. Measured: the model
|
|
546
|
+
// narrates on about 63% of steps and patches on the rest, so the
|
|
547
|
+
// state grows at roughly a third of the step rate — 17 files in 50
|
|
548
|
+
// steps, which puts 30 files at about 88 steps against a ceiling of
|
|
549
|
+
// 64. That is the whole of the 25/30.
|
|
550
|
+
maxSteps: maxStepsFromEnv(),
|
|
551
|
+
})
|
|
552
|
+
: undefined;
|
|
553
|
+
const maxSteps = runtime === undefined ? undefined : (maxStepsFromEnv() ?? DEFAULT_MAX_STEPS);
|
|
554
|
+
// Paper mode registers NO skillstate tools, and the reason is measured
|
|
555
|
+
// rather than doctrinal.
|
|
556
|
+
//
|
|
557
|
+
// `skillstate_update` is free-form by design — it cannot see the spec — so
|
|
558
|
+
// in paper mode it was a second write path into Sigma that bypasses
|
|
559
|
+
// `validatePatch` entirely. A thirty-file run left `total: '1523'` in the
|
|
560
|
+
// state file: a STRING, in a field the spec declares as `number`. No
|
|
561
|
+
// validated patch can produce that, so the model had used the tool, and
|
|
562
|
+
// nothing in the runtime noticed.
|
|
563
|
+
//
|
|
564
|
+
// §6.4's rollback guarantee is that "a rejected patch has no path into
|
|
565
|
+
// Sigma ... there is nothing to undo because there is nothing partially
|
|
566
|
+
// applied". An unvalidated second writer is exactly such a path. And the
|
|
567
|
+
// model does not need the tool to read: paper mode puts Sigma in the
|
|
568
|
+
// prompt by construction, which is the whole of eq. 1.
|
|
101
569
|
await ctx.tool.transform((editor) => {
|
|
102
|
-
|
|
570
|
+
// Notes mode gets the schema when the project SHIPPED one, so the tool
|
|
571
|
+
// that writes this file validates against the same §6.2 the paper's
|
|
572
|
+
// runtime uses. Only `source === 'file'` counts: a builtin spec is our own
|
|
573
|
+
// fallback, and holding a project's notes to it would reject notes that
|
|
574
|
+
// are fine. This is the same gate as `declaredFields`, for the same
|
|
575
|
+
// reason — a default is not a declaration.
|
|
576
|
+
if (mode !== 'paper') {
|
|
577
|
+
registerTools(editor, {
|
|
578
|
+
store,
|
|
579
|
+
sessions,
|
|
580
|
+
scopeFor,
|
|
581
|
+
// Every write resets the drift counter, in every mode. The paper-mode
|
|
582
|
+
// sink resets it from the event stream; this is the notes-mode path,
|
|
583
|
+
// and without it the notice is a false statement for the whole of
|
|
584
|
+
// notes mode — the state is on disk and the counter never learns it.
|
|
585
|
+
onWrite: (scope) => {
|
|
586
|
+
turnsSinceWrite.set(scope, 0);
|
|
587
|
+
stateWrites.set(scope, (stateWrites.get(scope) ?? 0) + 1);
|
|
588
|
+
},
|
|
589
|
+
...(resolution.source === 'file' && spec !== undefined ? { schema: spec.schema } : {}),
|
|
590
|
+
});
|
|
591
|
+
}
|
|
103
592
|
});
|
|
104
|
-
// ── Session tree
|
|
593
|
+
// ── Session tree and the paper-mode state sink ───────────────────────
|
|
105
594
|
// Sub-agent sessions are created by OpenCode itself, so the parent edge
|
|
106
595
|
// arrives on the event stream. Until one is seen a session is treated as
|
|
107
596
|
// a root session, which is the correct default for single-session use.
|
|
597
|
+
//
|
|
598
|
+
// The same stream carries the completed assistant text blocks that close
|
|
599
|
+
// the paper's transition, so both consumers share one subscription: a
|
|
600
|
+
// second `subscribe()` would be a second socket for no gain.
|
|
108
601
|
const controller = new AbortController();
|
|
109
602
|
void (async () => {
|
|
110
603
|
try {
|
|
111
604
|
for await (const event of ctx.event.subscribe({ signal: controller.signal })) {
|
|
112
|
-
|
|
605
|
+
try {
|
|
606
|
+
recordEvent(process.env['SKILLSTATE_DEBUG_PROMPT'], event.type);
|
|
607
|
+
sessions.ingestEvent(event);
|
|
608
|
+
// A sink failure is a value, never a throw — an unhandled
|
|
609
|
+
// rejection here would end the loop and silently stop both the
|
|
610
|
+
// registry and the sink for the rest of the process's life.
|
|
611
|
+
//
|
|
612
|
+
// The outcome is NOT discarded. Every rejection reason is queued as
|
|
613
|
+
// corrective feedback for the next prompt, because a model whose
|
|
614
|
+
// patch was refused and never told about it would otherwise be
|
|
615
|
+
// re-shown the identical context and keep failing silently.
|
|
616
|
+
const outcome = sink === undefined ? undefined : await sink.ingest(event);
|
|
617
|
+
if (outcome !== undefined && feedback !== undefined && isTextEnded(event)) {
|
|
618
|
+
feedback.record(event.data.sessionID, outcome);
|
|
619
|
+
}
|
|
620
|
+
if (outcome !== undefined && isTextEnded(event)) {
|
|
621
|
+
const key = scopeFor(event.data.sessionID);
|
|
622
|
+
if (outcome.applied) {
|
|
623
|
+
turnsSinceWrite.set(key, 0);
|
|
624
|
+
stateWrites.set(key, (stateWrites.get(key) ?? 0) + 1);
|
|
625
|
+
boundary.patchApplied(event.data.sessionID);
|
|
626
|
+
// Remembered, not acted on: the turn is not over yet, and
|
|
627
|
+
// ordering the next step here is what made the continuation
|
|
628
|
+
// arrive against a request that was already superseded.
|
|
629
|
+
//
|
|
630
|
+
// `action` needs no guard — the parser refuses a response
|
|
631
|
+
// without one, so `applied` already implies it. The gate proved
|
|
632
|
+
// the guard was dead by refusing to let it be covered.
|
|
633
|
+
lastAction.set(event.data.sessionID, outcome.action);
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
// ── The runtime owns the step, and a step is not a patch ───────
|
|
637
|
+
//
|
|
638
|
+
// §5.1, lines 9–10: if no valid patch was produced, the step
|
|
639
|
+
// returns (Σₜ, __invalid_patch__, {invalidated: true}) — the
|
|
640
|
+
// state is UNCHANGED and the loop continues anyway. Advancing only
|
|
641
|
+
// on `applied` therefore deleted the failure case: a turn that
|
|
642
|
+
// produced prose instead of a patch ended the procedure, when the
|
|
643
|
+
// paper says it should have been retried with the reason attached.
|
|
644
|
+
// The feedback queue carries that reason; it just never got reached,
|
|
645
|
+
// because the loop stopped before the next turn.
|
|
646
|
+
//
|
|
647
|
+
// Fired on `session.step.ended`, NOT on a completed text block. A text block ends
|
|
648
|
+
// when the model's response ends, which is not the same thing: a
|
|
649
|
+
// model that narrates and then calls a tool has ended a text block
|
|
650
|
+
// and is nowhere near done. Advancing there ordered the next step
|
|
651
|
+
// while the current one was still running, and the continuation was
|
|
652
|
+
// consumed by a request that got superseded — measured, the
|
|
653
|
+
// `[next step]` marker never reached the model at all.
|
|
654
|
+
if (mode === 'paper' && isStepEnded(event)) {
|
|
655
|
+
const sessionID = event.data.sessionID;
|
|
656
|
+
const last = lastAction.get(sessionID);
|
|
657
|
+
// §5.1 lines 2–8: one step is `k + 1` attempts at the SAME Aₜ.
|
|
658
|
+
// A turn that produced no patch is an attempt, not a step, so the
|
|
659
|
+
// corrective feedback arrives on the prompt it belongs to instead
|
|
660
|
+
// of on the next step's entirely different one.
|
|
661
|
+
const verdict = runtime === undefined
|
|
662
|
+
? undefined
|
|
663
|
+
: runtime.record(sessionID, last !== undefined);
|
|
664
|
+
if (verdict?.result === INVALID_PATCH) {
|
|
665
|
+
invalidations.set(sessionID, {
|
|
666
|
+
attempts: verdict.attempt,
|
|
667
|
+
lastError: feedback?.peek(sessionID),
|
|
668
|
+
});
|
|
669
|
+
}
|
|
670
|
+
if (runtime !== undefined) {
|
|
671
|
+
// Deferred out of the event loop: asking the server to start a
|
|
672
|
+
// turn from inside the handler reporting that turn is re-entrant,
|
|
673
|
+
// and the request is dropped.
|
|
674
|
+
setTimeout(() => {
|
|
675
|
+
void runtime
|
|
676
|
+
?.advance(sessionID, last ?? CONTINUE_ACTION)
|
|
677
|
+
.then((step) => {
|
|
678
|
+
// A `null` here is the loop ending, and four different
|
|
679
|
+
// things end it. `max_steps` is the one that lies — a loop
|
|
680
|
+
// stopped at the ceiling looks exactly like a loop that ran
|
|
681
|
+
// its course, and that is how a 90-file run got read as a
|
|
682
|
+
// model that lost track of its sum at file 78 when the
|
|
683
|
+
// ceiling had arrived mid-file-79. Written next to the
|
|
684
|
+
// build stamp so a scorer never has to guess it from a
|
|
685
|
+
// transcript.
|
|
686
|
+
const stop = runtime?.lastStop;
|
|
687
|
+
// Both arguments are non-optional, and that is deliberate:
|
|
688
|
+
// this is only reached when `advance` returned null, which
|
|
689
|
+
// is only reached when a runtime exists, so `lastStop` is
|
|
690
|
+
// always set and the ceiling was resolved when the runtime
|
|
691
|
+
// was built. The `?.` and the `| undefined` were branches
|
|
692
|
+
// nothing could take.
|
|
693
|
+
if (step === null && stop !== undefined && maxSteps !== undefined) {
|
|
694
|
+
writeRunRecord(directory, stop, maxSteps);
|
|
695
|
+
}
|
|
696
|
+
});
|
|
697
|
+
}, 0);
|
|
698
|
+
}
|
|
699
|
+
// What the loop did, step by step. Read BEFORE the action is
|
|
700
|
+
// forgotten, and after the state is written, so the line answers
|
|
701
|
+
// the only question that matters here: did the turn before this
|
|
702
|
+
// one produce a patch, and what did the state say afterwards?
|
|
703
|
+
const snapshot = store.read(scopeFor(sessionID));
|
|
704
|
+
dumpStepTrace(process.env['SKILLSTATE_DEBUG_STEPS'], {
|
|
705
|
+
sessionID,
|
|
706
|
+
step: verdict?.step ?? -1,
|
|
707
|
+
attempt: verdict?.attempt ?? 0,
|
|
708
|
+
applied: last !== undefined,
|
|
709
|
+
done: Array.isArray(snapshot?.done) ? snapshot.done.length : -1,
|
|
710
|
+
total: typeof snapshot?.total === 'number' ? snapshot.total : null,
|
|
711
|
+
drove: runtime !== undefined,
|
|
712
|
+
note: last ?? CONTINUE_ACTION,
|
|
713
|
+
});
|
|
714
|
+
lastAction.delete(sessionID);
|
|
715
|
+
}
|
|
716
|
+
}
|
|
717
|
+
catch {
|
|
718
|
+
// Per EVENT, not per stream. The outer catch ends the loop, and the
|
|
719
|
+
// loop is the only source of the registry and the sink — so one
|
|
720
|
+
// malformed event used to end both for the lifetime of the process,
|
|
721
|
+
// silently. Found by a test that fed the loop a bare `null`.
|
|
722
|
+
}
|
|
113
723
|
}
|
|
114
724
|
}
|
|
115
725
|
catch {
|
|
@@ -118,25 +728,182 @@ export const SkillStatePlugin = Plugin.define({
|
|
|
118
728
|
// which is safe; it must never surface as an unhandled rejection.
|
|
119
729
|
}
|
|
120
730
|
})();
|
|
121
|
-
// ──
|
|
731
|
+
// ── Context ─────────────────────────────────────────────────────────
|
|
122
732
|
// Registered on the agent loop only. `compaction`, `generate` and
|
|
123
733
|
// `title` are separate hooks in v2 and are deliberately left alone:
|
|
124
|
-
// after a compaction the next agent-loop request re-
|
|
125
|
-
//
|
|
126
|
-
// the
|
|
127
|
-
|
|
734
|
+
// after a compaction the next agent-loop request re-enters here, so
|
|
735
|
+
// state survives without this plugin ever touching the summariser's
|
|
736
|
+
// input. The same holds in paper mode, where the compaction summary is
|
|
737
|
+
// discarded along with the rest of the transcript.
|
|
738
|
+
await ctx.session.hook('context', async (event) => {
|
|
128
739
|
const scope = scopeFor(event.sessionID);
|
|
740
|
+
// Read-after-write against the host. The patch the model just emitted is
|
|
741
|
+
// already in this transcript, and the host does not wait for the event
|
|
742
|
+
// loop to deliver it, so waiting for `session.text.ended` serves the
|
|
743
|
+
// next request a Sigma that has not moved. Recovering it here is what
|
|
744
|
+
// makes the state the model is shown match the state on disk.
|
|
745
|
+
const recovered = await sink?.recover(event.sessionID, event.messages);
|
|
746
|
+
// A patch recovered here is an applied patch, so the model has accounted
|
|
747
|
+
// for its step and may act again. Leaving the phase alone would demand a
|
|
748
|
+
// second report turn for the same patch — the model would be told to
|
|
749
|
+
// account for something it had already reported.
|
|
750
|
+
if (recovered?.applied === true) {
|
|
751
|
+
boundary.patchApplied(event.sessionID);
|
|
752
|
+
}
|
|
753
|
+
const state = store.read(scope);
|
|
754
|
+
if (mode === 'paper') {
|
|
755
|
+
// ── §5.1's step boundary ──────────────────────────────────────────
|
|
756
|
+
// One request may act; the next must account for it. `tools` is
|
|
757
|
+
// handed to this hook on every model request, so the cycle is
|
|
758
|
+
// enforced by withholding the tools rather than by asking in prose.
|
|
759
|
+
// See step-boundary.ts for why delegation to the host's agent loop
|
|
760
|
+
// is not the same thing.
|
|
761
|
+
const target = event;
|
|
762
|
+
// A tool result in the incoming transcript means the host has just
|
|
763
|
+
// executed an action for this session. That is the observable edge of
|
|
764
|
+
// §5.1's `execute(aₜ, Σₜ₊₁)`, and it is what moves the session from
|
|
765
|
+
// `act` to `report`.
|
|
766
|
+
//
|
|
767
|
+
// OFF BY DEFAULT, and the reason is a measurement rather than a
|
|
768
|
+
// preference. The premise — that a model asked again with no tools
|
|
769
|
+
// available can only answer in text, and that text is where a patch
|
|
770
|
+
// lives — is false for the models tested here. Measured on the
|
|
771
|
+
// eight-file task with the boundary on: two text blocks, NEITHER
|
|
772
|
+
// containing a `state_patch`, and the model writing prose instead
|
|
773
|
+
// ("the saved execution state is still {total:0, files:0}… I will
|
|
774
|
+
// restart from src/cfg1.ts"). It had done the accounting it was asked
|
|
775
|
+
// for, in words, and Σ never moved. With the boundary off the same
|
|
776
|
+
// task reaches cfg8; with it on it stops at cfg1.
|
|
777
|
+
//
|
|
778
|
+
// So the mechanism is kept, correct and tested, behind
|
|
779
|
+
// SKILLSTATE_STEP_BOUNDARY=1, and the default stays the behaviour that
|
|
780
|
+
// measurably goes further. Turning it on is a claim to be measured, not
|
|
781
|
+
// a setting to leave flipped.
|
|
782
|
+
// Two different questions, deliberately two different answers.
|
|
783
|
+
//
|
|
784
|
+
// Did an action run? That is the boundary's question and it is about the
|
|
785
|
+
// turn, so it asks the turn — `hasToolResult` looks at the newest message
|
|
786
|
+
// and does not care whether this hook has seen it before.
|
|
787
|
+
//
|
|
788
|
+
// How many ran, for the report? That is the driver's question and it has
|
|
789
|
+
// to be about *new* results: this hook runs on every model request, so a
|
|
790
|
+
// tool message still newest across two requests would be counted twice,
|
|
791
|
+
// and "2 actions ran" for one `read` is exactly the kind of lie the
|
|
792
|
+
// report exists to stop. So the driver keeps the watermark and the
|
|
793
|
+
// caller does not have to know it exists.
|
|
794
|
+
if (hasToolResult(target.messages))
|
|
795
|
+
boundary.actionTaken(event.sessionID);
|
|
796
|
+
runtime?.noteExecutions(event.sessionID, target.messages);
|
|
797
|
+
if (stepBoundaryEnabled && boundary.reportRequired(event.sessionID)) {
|
|
798
|
+
target.tools = {};
|
|
799
|
+
}
|
|
800
|
+
// A session that has saved nothing yet has no Σₜ to show, and
|
|
801
|
+
// replacing the context with an empty state block before the agent
|
|
802
|
+
// has done anything would only lose the task. Stay inert.
|
|
803
|
+
//
|
|
804
|
+
// Note the feedback is deliberately NOT taken here: this early return
|
|
805
|
+
// happens before any prompt is built, so consuming the correction
|
|
806
|
+
// would drop it without ever showing it to the model.
|
|
807
|
+
if (Object.keys(state).length === 0)
|
|
808
|
+
return;
|
|
809
|
+
// Taken exactly once: `take` clears on read, so calling it twice would
|
|
810
|
+
// show the correction to nobody.
|
|
811
|
+
// §6.4: a step that spent all `k + 1` attempts produces a synthetic
|
|
812
|
+
// observation instead of the running correction, and the sentinel action
|
|
813
|
+
// is never executed. The invalidation is recorded by `record` on the
|
|
814
|
+
// turn that exhausted it, so this is where the model finally hears it.
|
|
815
|
+
const invalidation = invalidations.get(event.sessionID);
|
|
816
|
+
invalidations.delete(event.sessionID);
|
|
817
|
+
const correction = invalidation === undefined
|
|
818
|
+
? feedback?.take(event.sessionID)
|
|
819
|
+
: feedback?.takeUnlessInvalidated(event.sessionID, invalidation.attempts, invalidation.lastError);
|
|
820
|
+
// The action the runtime is carrying out, which has to reach the model
|
|
821
|
+
// through Oₜ because the messages it was sent in are cleared here.
|
|
822
|
+
// What rides in Oₜ, and `SKILLSTATE_CONTINUATION` chooses between three
|
|
823
|
+
// things, because both extremes have been measured and both are wrong:
|
|
824
|
+
//
|
|
825
|
+
// unset (default) the environment's REPORT of what the runtime did
|
|
826
|
+
// '1' the action the model itself proposed, as an order
|
|
827
|
+
// '0' nothing at all
|
|
828
|
+
//
|
|
829
|
+
// §2 forbids the middle one — "the agent receives only Oₜ, never prior
|
|
830
|
+
// observations or actions" — and with it the model obeyed its own stored
|
|
831
|
+
// order: 54 reads for thirty files where a control used one grep. The
|
|
832
|
+
// empty end loses too: the runtime re-prompts when a turn produced a
|
|
833
|
+
// patch but no tool call, and with nothing saying why, the model looped
|
|
834
|
+
// — 98 text blocks against 43, 7.9M tokens against 1.6M. See
|
|
835
|
+
// `RuntimeDriver.stepReport`.
|
|
836
|
+
const continuationFlag = process.env['SKILLSTATE_CONTINUATION'];
|
|
837
|
+
const [continuationText, continuationKind] = continuationFlag === '0'
|
|
838
|
+
? [undefined, 'report']
|
|
839
|
+
: continuationFlag === '1'
|
|
840
|
+
? [runtime?.takeContinuation(event.sessionID) ?? '', 'order']
|
|
841
|
+
: [runtime?.stepReport(event.sessionID) ?? '', 'report'];
|
|
842
|
+
const raw = event.messages;
|
|
843
|
+
dumpPromptShape(process.env['SKILLSTATE_DEBUG_PROMPT'], raw, state);
|
|
844
|
+
applyPaperContext(event, buildPaperPrompt({
|
|
845
|
+
spec: spec,
|
|
846
|
+
state,
|
|
847
|
+
messages: raw,
|
|
848
|
+
// §2, on Observation: "The agent receives only Oₜ — never prior
|
|
849
|
+
// observations or ACTIONS."
|
|
850
|
+
//
|
|
851
|
+
// That is not an interpretation and it is not a preference. Putting
|
|
852
|
+
// the model's own previous action into Oₜ is putting an action into
|
|
853
|
+
// the channel the paper reserves for the environment's reply, and
|
|
854
|
+
// this implementation did it to close a real gap — the model was
|
|
855
|
+
// re-prompted with nothing saying why.
|
|
856
|
+
//
|
|
857
|
+
// It worked too well, which is how it was found. The model began
|
|
858
|
+
// answering "I'll read cfg3.ts next, as directed by the
|
|
859
|
+
// observation" and then reading cfg3.ts. It obeyed a stored order
|
|
860
|
+
// instead of choosing, made 54 `read` calls for thirty files, and
|
|
861
|
+
// used grep three times as a side errand. A control with no step
|
|
862
|
+
// driver read one file, ran ONE grep and finished in six calls. The
|
|
863
|
+
// order was its own past action, so it never looked for a better
|
|
864
|
+
// way than the one it had already written down.
|
|
865
|
+
...(continuationText === undefined || continuationText.length === 0
|
|
866
|
+
? {}
|
|
867
|
+
: { continuation: continuationText, continuationKind }),
|
|
868
|
+
...(correction === undefined ? {} : { feedback: correction }),
|
|
869
|
+
}), HOST_ACTION_NOTE);
|
|
870
|
+
return;
|
|
871
|
+
}
|
|
129
872
|
if (!store.exists(scope))
|
|
130
873
|
return;
|
|
131
|
-
|
|
874
|
+
// ── Drift detection ───────────────────────────────────────────────
|
|
875
|
+
// A user who initialized skillstate did so because the work needs
|
|
876
|
+
// cross-turn memory. An agent that then quietly stops writing drifts
|
|
877
|
+
// back to a growing transcript and pays for it in re-sent tokens. The
|
|
878
|
+
// fix is to notice and say so — a measured fact, not an instruction,
|
|
879
|
+
// so it cannot displace the task the way the v1 injection did.
|
|
880
|
+
//
|
|
881
|
+
// Counted per scope and reset by the sink on every applied patch, so
|
|
882
|
+
// a sub-agent's writes do not silence the main session's counter.
|
|
883
|
+
const sinceWrite = (turnsSinceWrite.get(scope) ?? 0) + 1;
|
|
884
|
+
turnsSinceWrite.set(scope, sinceWrite);
|
|
132
885
|
const hint = buildStateHint({
|
|
133
886
|
state,
|
|
134
887
|
statePath: path.relative(store.projectDirectory, store.pathFor(scope)),
|
|
135
888
|
scope,
|
|
889
|
+
declaredFields,
|
|
890
|
+
// A state file on disk is the definition of an initialized project:
|
|
891
|
+
// the user ran `skillstate init`, or something wrote one.
|
|
892
|
+
initialized: store.exists(scope),
|
|
893
|
+
turnsSinceWrite: sinceWrite,
|
|
136
894
|
});
|
|
137
895
|
if (hint.length === 0)
|
|
138
896
|
return;
|
|
139
897
|
event.system.push({ type: 'text', text: hint });
|
|
898
|
+
// What went out, and what the model had done about it at the time. The
|
|
899
|
+
// only way to tell "the notice was ignored" from "the notice was never
|
|
900
|
+
// built" — the two are indistinguishable from the outside otherwise.
|
|
901
|
+
dumpDrift(process.env['SKILLSTATE_DEBUG_DRIFT'], {
|
|
902
|
+
scope,
|
|
903
|
+
turns: sinceWrite,
|
|
904
|
+
notice: hint.includes(driftNotice(sinceWrite)),
|
|
905
|
+
writes: stateWrites.get(scope) ?? 0,
|
|
906
|
+
});
|
|
140
907
|
});
|
|
141
908
|
return () => {
|
|
142
909
|
controller.abort();
|