@moxxy/reflector-default 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.ts ADDED
@@ -0,0 +1,428 @@
1
+ import {
2
+ definePlugin,
3
+ z,
4
+ type LifecycleHooks,
5
+ type LLMProvider,
6
+ type MoxxyEvent,
7
+ type Plugin,
8
+ type ProviderRequest,
9
+ type ReflectContext,
10
+ type ReflectionProposal,
11
+ type ReflectorDef,
12
+ type SessionId,
13
+ } from '@moxxy/sdk';
14
+
15
+ /**
16
+ * @moxxy/reflector-default — the default learning-loop block.
17
+ *
18
+ * TWO parts ship in one plugin:
19
+ * 1. a {@link ReflectorDef} named `'default'` (the strategy): one cheap
20
+ * side-channel LLM pass over a finished turn that returns 0-2
21
+ * {@link ReflectionProposal}s;
22
+ * 2. the DRIVER (lifecycle hooks): decides WHEN to reflect (a cheap gate +
23
+ * a one-per-session budget), runs the active reflector fire-and-forget on
24
+ * `onTurnEnd`, and delivers any proposals as a ONE-TIME nudge on the next
25
+ * `onBeforeProviderCall`.
26
+ *
27
+ * The load-bearing property is "propose, don't write": a proposal never mutates
28
+ * memory or skills. It is injected into the next request's system prompt phrased
29
+ * so the MODEL may call `memory_save` / `synthesize_skill` — which still hit
30
+ * their own permission prompts. The reflection itself is best-effort: it runs
31
+ * detached from the turn, catches everything, and skips silently when there is
32
+ * no provider or the provider errors.
33
+ */
34
+
35
+ // ── Gate thresholds ─────────────────────────────────────────────────────────
36
+ /** Reflect when a turn did at least this many tool calls (busy/procedural turn). */
37
+ export const REFLECT_MIN_TOOL_RESULTS = 5;
38
+ /** Reflect when a turn ran at least this many mode-loop rounds (a hard task). */
39
+ export const REFLECT_MIN_ITERATIONS = 8;
40
+ /** Upper bound on the whole reflection (side-channel LLM pass). */
41
+ export const REFLECT_TIMEOUT_MS = 30_000;
42
+ /** Cap on the model reply span we'll JSON-parse — past this is malformed/hostile. */
43
+ const MAX_JSON_SPAN = 32 * 1024;
44
+ /** Cap on the turn digest handed to the reflection model. */
45
+ const MAX_DIGEST_CHARS = 4000;
46
+ /** Cap on a single snippet (prompt / assistant text / one error) inside the digest. */
47
+ const MAX_SNIPPET_CHARS = 600;
48
+ /** Tokens for the reflection pass — proposals are tiny. */
49
+ const REFLECT_MAX_TOKENS = 1200;
50
+
51
+ export const REFLECT_SYSTEM_PROMPT = `You review a single finished agent turn and decide whether anything is worth remembering for the future.
52
+
53
+ Look for exactly two things:
54
+ - a durable FACT worth saving to long-term memory (a stable preference, a project detail, a hard-won answer) → kind "memory"
55
+ - a repeated multi-step PROCEDURE worth capturing as a reusable skill → kind "skill"
56
+
57
+ Be strict. Most turns warrant NOTHING. Never propose transient chit-chat, one-off values, or things already obvious from context.
58
+
59
+ Output ONLY a JSON array with 0, 1, or 2 items — no prose, no code fences. Each item:
60
+ { "kind": "memory" | "skill", "title": "<= 80 chars", "nudge": "one paragraph addressed to the assistant, suggesting it consider saving this" }
61
+ Return [] when nothing is worth proposing.`;
62
+
63
+ // ── Pure helpers (exported for tests) ───────────────────────────────────────
64
+
65
+ /**
66
+ * The cheap gate, run synchronously on `onTurnEnd` against the turn's events.
67
+ * A turn is worth reflecting on when it was busy (≥{@link REFLECT_MIN_TOOL_RESULTS}
68
+ * tool results), hit an error, or ground through ≥{@link REFLECT_MIN_ITERATIONS}
69
+ * mode-loop rounds. A quiet turn (a quick Q&A) is skipped.
70
+ */
71
+ export function shouldReflect(events: ReadonlyArray<MoxxyEvent>): boolean {
72
+ let toolResults = 0;
73
+ let errors = 0;
74
+ let iterations = 0;
75
+ for (const e of events) {
76
+ if (e.type === 'tool_result') toolResults++;
77
+ else if (e.type === 'error') errors++;
78
+ else if (e.type === 'mode_iteration') iterations++;
79
+ }
80
+ return (
81
+ toolResults >= REFLECT_MIN_TOOL_RESULTS || errors >= 1 || iterations >= REFLECT_MIN_ITERATIONS
82
+ );
83
+ }
84
+
85
+ function clip(text: string, max: number): string {
86
+ const t = text.trim();
87
+ return t.length > max ? t.slice(0, max) + '…' : t;
88
+ }
89
+
90
+ /**
91
+ * Build a compact text digest of a turn from its events: the user prompt, the
92
+ * tool names invoked (with counts), any error snippets, and the final assistant
93
+ * text — each snippet clipped, the whole thing capped at {@link MAX_DIGEST_CHARS}.
94
+ * This is what the reflection model reads instead of the raw event stream.
95
+ */
96
+ export function buildTurnDigest(events: ReadonlyArray<MoxxyEvent>): string {
97
+ const parts: string[] = [];
98
+
99
+ const prompt = events.find((e) => e.type === 'user_prompt');
100
+ if (prompt && prompt.type === 'user_prompt') {
101
+ parts.push(`USER: ${clip(prompt.text, MAX_SNIPPET_CHARS)}`);
102
+ }
103
+
104
+ const toolCounts = new Map<string, number>();
105
+ for (const e of events) {
106
+ if (e.type === 'tool_call_requested') {
107
+ toolCounts.set(e.name, (toolCounts.get(e.name) ?? 0) + 1);
108
+ }
109
+ }
110
+ if (toolCounts.size > 0) {
111
+ const list = [...toolCounts.entries()]
112
+ .map(([name, n]) => (n > 1 ? `${name}×${n}` : name))
113
+ .join(', ');
114
+ parts.push(`TOOLS: ${list}`);
115
+ }
116
+
117
+ const errorSnippets: string[] = [];
118
+ for (const e of events) {
119
+ if (e.type === 'tool_result' && e.ok === false && e.error?.message) {
120
+ errorSnippets.push(clip(e.error.message, 200));
121
+ } else if (e.type === 'error') {
122
+ errorSnippets.push(clip(e.message, 200));
123
+ }
124
+ }
125
+ if (errorSnippets.length > 0) {
126
+ parts.push(`ERRORS: ${errorSnippets.slice(0, 5).join(' | ')}`);
127
+ }
128
+
129
+ // Last finalized assistant message is the turn's conclusion.
130
+ let finalText: string | null = null;
131
+ for (const e of events) {
132
+ if (e.type === 'assistant_message' && e.content.trim()) finalText = e.content;
133
+ }
134
+ if (finalText) parts.push(`ASSISTANT: ${clip(finalText, MAX_SNIPPET_CHARS)}`);
135
+
136
+ return clip(parts.join('\n'), MAX_DIGEST_CHARS);
137
+ }
138
+
139
+ const proposalSchema = z.object({
140
+ kind: z.enum(['memory', 'skill']),
141
+ title: z.string().min(1).max(200),
142
+ nudge: z.string().min(1).max(2000),
143
+ });
144
+ const proposalsSchema = z.array(proposalSchema);
145
+
146
+ /**
147
+ * Defensively parse the reflection model's reply into 0-2 proposals. Strips an
148
+ * optional ```json fence, takes the first `[` … last `]` span, refuses an
149
+ * oversized span, JSON-parses, validates each item, and clamps to 2. ANY
150
+ * failure (no array, malformed JSON, wrong shape) yields `[]` — a bad reply
151
+ * silently produces no nudge rather than throwing into the detached reflection.
152
+ */
153
+ export function parseReflectionReply(text: string): ReflectionProposal[] {
154
+ try {
155
+ const fenced = /```(?:json)?\n?([\s\S]*?)```/.exec(text);
156
+ const candidate = fenced ? fenced[1]! : text;
157
+ const start = candidate.indexOf('[');
158
+ const end = candidate.lastIndexOf(']');
159
+ if (start === -1 || end <= start) return [];
160
+ const span = candidate.slice(start, end + 1);
161
+ if (span.length > MAX_JSON_SPAN) return [];
162
+ const parsed = proposalsSchema.safeParse(JSON.parse(span));
163
+ if (!parsed.success) return [];
164
+ return parsed.data.slice(0, 2).map((p) => ({ kind: p.kind, title: p.title, nudge: p.nudge }));
165
+ } catch {
166
+ return [];
167
+ }
168
+ }
169
+
170
+ /**
171
+ * The one-time nudge injected into the next provider request's system prompt.
172
+ * Phrased as a suggestion the model MAY act on via `memory_save` /
173
+ * `synthesize_skill` (each of which asks the user for permission) — never a
174
+ * directive, never a silent write.
175
+ */
176
+ export function buildNudgeBlock(proposals: ReadonlyArray<ReflectionProposal>): string {
177
+ const lines = proposals.map((p) => `- (${p.kind}) ${p.title}: ${p.nudge}`).join('\n');
178
+ return (
179
+ `\n\n[reflection] A background review of the last turn surfaced ${proposals.length} ` +
180
+ `suggestion(s) you MAY act on if genuinely useful (otherwise ignore):\n${lines}\n` +
181
+ 'To persist one, you may call `memory_save` (a durable fact) or `synthesize_skill` ' +
182
+ '(a repeatable procedure) — each asks the user for permission first. Do not act unless it clearly helps.'
183
+ );
184
+ }
185
+
186
+ // ── The `default` reflector (the strategy) ──────────────────────────────────
187
+
188
+ /** Minimal view of the provider registry the reflection pass needs. */
189
+ interface ActiveProviderSource {
190
+ getActiveName(): string | null;
191
+ getActive(): LLMProvider;
192
+ }
193
+
194
+ /** Collect a provider stream into text, honoring the abort signal. */
195
+ async function streamText(provider: LLMProvider, req: ProviderRequest): Promise<string> {
196
+ let out = '';
197
+ const iterable = provider.stream(req);
198
+ const iterator = iterable[Symbol.asyncIterator]();
199
+ try {
200
+ for (;;) {
201
+ if (req.signal?.aborted) break;
202
+ const step = await raceAbort(iterator.next(), req.signal);
203
+ if (step === ABORTED) break;
204
+ if (step.done) break;
205
+ const event = step.value;
206
+ if (event.type === 'text_delta') out += event.delta;
207
+ else if (event.type === 'error') throw new Error(event.message);
208
+ }
209
+ } finally {
210
+ if (req.signal?.aborted) void iterator.return?.(undefined).catch(() => {});
211
+ }
212
+ return out;
213
+ }
214
+
215
+ const ABORTED = Symbol('aborted');
216
+ function raceAbort<T>(step: Promise<T>, signal: AbortSignal | undefined): Promise<T | typeof ABORTED> {
217
+ if (!signal) return step;
218
+ if (signal.aborted) {
219
+ void step.catch(() => {});
220
+ return Promise.resolve(ABORTED);
221
+ }
222
+ return new Promise<T | typeof ABORTED>((resolve, reject) => {
223
+ const onAbort = () => {
224
+ void step.catch(() => {});
225
+ resolve(ABORTED);
226
+ };
227
+ signal.addEventListener('abort', onAbort, { once: true });
228
+ step.then(
229
+ (v) => {
230
+ signal.removeEventListener('abort', onAbort);
231
+ resolve(v);
232
+ },
233
+ (e) => {
234
+ signal.removeEventListener('abort', onAbort);
235
+ reject(e);
236
+ },
237
+ );
238
+ });
239
+ }
240
+
241
+ /**
242
+ * The default reflector: one cheap side-channel LLM pass over the finished
243
+ * turn. Resolves the active provider from the service registry, streams a
244
+ * strict-JSON reply, and parses it into 0-2 proposals. Returns `[]` (never
245
+ * throws) when there is no active provider or the provider errors — reflection
246
+ * degrades to a no-op rather than surfacing anything into the session.
247
+ */
248
+ export const reflectorDefaultDef: ReflectorDef = {
249
+ name: 'default',
250
+ displayName: 'Default learning loop',
251
+ async reflect(ctx: ReflectContext): Promise<ReadonlyArray<ReflectionProposal>> {
252
+ const providers = ctx.services.get<ActiveProviderSource>('providers');
253
+ // Graceful no-provider skip: nothing to reflect with.
254
+ if (!providers || providers.getActiveName() == null) return [];
255
+ let provider: LLMProvider;
256
+ try {
257
+ provider = providers.getActive();
258
+ } catch {
259
+ return [];
260
+ }
261
+
262
+ const digest = buildTurnDigest(ctx.log.byTurn(ctx.turnId));
263
+ if (!digest) return [];
264
+
265
+ const req: ProviderRequest = {
266
+ model: provider.models[0]?.id ?? 'unknown',
267
+ system: REFLECT_SYSTEM_PROMPT,
268
+ messages: [{ role: 'user', content: [{ type: 'text', text: digest }] }],
269
+ maxTokens: REFLECT_MAX_TOKENS,
270
+ signal: ctx.signal,
271
+ };
272
+ try {
273
+ const reply = await streamText(provider, req);
274
+ return parseReflectionReply(reply);
275
+ } catch {
276
+ // Provider error / abort → degrade to a no-op rather than surfacing.
277
+ return [];
278
+ }
279
+ },
280
+ };
281
+
282
+ // ── The driver (lifecycle hooks) ────────────────────────────────────────────
283
+
284
+ /** Per-session reflection budget: window = the whole session for v1. */
285
+ export interface SessionBudget {
286
+ /** Reflections fired this window (v1 cap = 1). */
287
+ readonly count: number;
288
+ /** Seq of the last turn's last event reflected on (future windowing aid). */
289
+ readonly lastSeq: number;
290
+ }
291
+
292
+ /** Minimal view of the reflector registry the driver resolves the active def from. */
293
+ interface ActiveReflectorSource {
294
+ getActive(): ReflectorDef | null;
295
+ }
296
+
297
+ /** Test-facing handle into the driver's private per-session state. */
298
+ export interface ReflectorInternals {
299
+ /** Await the (fire-and-forget) reflection kicked off for a session, if any. */
300
+ settle(sessionId: SessionId): Promise<void>;
301
+ /** The pending one-time nudge for a session, or null. */
302
+ pendingNudge(sessionId: SessionId): string | null;
303
+ /** The session's reflection budget, or undefined when it never fired the gate. */
304
+ budget(sessionId: SessionId): SessionBudget | undefined;
305
+ }
306
+
307
+ export interface BuildReflectorPluginResult {
308
+ readonly plugin: Plugin;
309
+ readonly internals: ReflectorInternals;
310
+ }
311
+
312
+ /**
313
+ * Build the reflector plugin. The default export calls this with no options;
314
+ * tests use the returned `internals` handle to await the detached reflection
315
+ * and inspect the budget / pending nudge deterministically.
316
+ */
317
+ export function buildReflectorPlugin(): BuildReflectorPluginResult {
318
+ // All keyed by sessionId so one shared instance stays correct across the many
319
+ // sessions a runner hosts in one process; every map self-cleans in onShutdown.
320
+ const budgets = new Map<SessionId, SessionBudget>();
321
+ const pending = new Map<SessionId, string>();
322
+ const inFlight = new Map<SessionId, Promise<void>>();
323
+ const shutdownControllers = new Map<SessionId, AbortController>();
324
+
325
+ function cleanup(sessionId: SessionId): void {
326
+ try {
327
+ shutdownControllers.get(sessionId)?.abort();
328
+ } catch {
329
+ // aborting a controller never throws in practice; guard anyway.
330
+ }
331
+ shutdownControllers.delete(sessionId);
332
+ budgets.delete(sessionId);
333
+ pending.delete(sessionId);
334
+ inFlight.delete(sessionId);
335
+ }
336
+
337
+ const hooks: LifecycleHooks = {
338
+ // AWAITED by the lifecycle dispatcher (with a per-plugin timeout), so this
339
+ // MUST return fast: the gate + budget check are synchronous, and the actual
340
+ // reflection is kicked off FIRE-AND-FORGET (a guarded detached promise) so
341
+ // it never blocks the turn and can never throw into it.
342
+ onTurnEnd(ctx) {
343
+ try {
344
+ const events = ctx.log.byTurn(ctx.turnId);
345
+ if (!shouldReflect(events)) return;
346
+
347
+ const prior = budgets.get(ctx.sessionId) ?? { count: 0, lastSeq: -1 };
348
+ // Budget: at most one reflection per session window (v1 = whole session).
349
+ if (prior.count >= 1) return;
350
+
351
+ const reflectors = ctx.services.get<ActiveReflectorSource>('reflectors');
352
+ const reflector = reflectors?.getActive() ?? null;
353
+ // Nullable registry: no active reflector → reflection is off. Do NOT
354
+ // consume the budget (so activating one later still gets a chance).
355
+ if (!reflector) return;
356
+
357
+ // Reserve the budget slot BEFORE going async so a rapid next turn can't
358
+ // double-fire while this reflection is still in flight.
359
+ const lastSeq = events.length > 0 ? events[events.length - 1]!.seq : prior.lastSeq;
360
+ budgets.set(ctx.sessionId, { count: prior.count + 1, lastSeq });
361
+
362
+ let sc = shutdownControllers.get(ctx.sessionId);
363
+ if (!sc) {
364
+ sc = new AbortController();
365
+ shutdownControllers.set(ctx.sessionId, sc);
366
+ }
367
+ const signal = AbortSignal.any([sc.signal, AbortSignal.timeout(REFLECT_TIMEOUT_MS)]);
368
+ const reflectCtx: ReflectContext = {
369
+ sessionId: ctx.sessionId,
370
+ turnId: ctx.turnId,
371
+ cwd: ctx.cwd,
372
+ log: ctx.log,
373
+ services: ctx.services,
374
+ signal,
375
+ };
376
+
377
+ const p = (async () => {
378
+ try {
379
+ const proposals = await reflector.reflect(reflectCtx);
380
+ if (proposals.length > 0) pending.set(ctx.sessionId, buildNudgeBlock(proposals));
381
+ } catch {
382
+ // Best-effort: a no-provider skip, provider error, or reflector
383
+ // throw must never surface. The budget stays spent (one attempt).
384
+ }
385
+ })();
386
+ inFlight.set(ctx.sessionId, p);
387
+ // Detach: do NOT return `p` — onTurnEnd resolves immediately.
388
+ void p;
389
+ } catch {
390
+ // The gate itself (a hostile log reader, etc.) must never take down the
391
+ // awaited onTurnEnd hook.
392
+ }
393
+ },
394
+
395
+ // Deliver a pending reflection as a ONE-TIME nudge on the next provider call.
396
+ onBeforeProviderCall(req, ctx) {
397
+ const nudge = pending.get(ctx.sessionId);
398
+ if (!nudge) return;
399
+ pending.delete(ctx.sessionId); // one-shot: cleared after injection
400
+ return { ...req, system: (req.system ?? '') + nudge };
401
+ },
402
+
403
+ onShutdown(ctx) {
404
+ cleanup(ctx.sessionId);
405
+ },
406
+ };
407
+
408
+ const plugin = definePlugin({
409
+ name: '@moxxy/reflector-default',
410
+ version: '0.0.0',
411
+ reflectors: [reflectorDefaultDef],
412
+ hooks,
413
+ });
414
+
415
+ const internals: ReflectorInternals = {
416
+ async settle(sessionId) {
417
+ await inFlight.get(sessionId)?.catch(() => {});
418
+ },
419
+ pendingNudge: (sessionId) => pending.get(sessionId) ?? null,
420
+ budget: (sessionId) => budgets.get(sessionId),
421
+ };
422
+
423
+ return { plugin, internals };
424
+ }
425
+
426
+ /** Discovery-loadable default export (the plain plugin, no test handle). */
427
+ const reflectorPlugin: Plugin = buildReflectorPlugin().plugin;
428
+ export default reflectorPlugin;