@mercury-fw/core 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +38 -0
- package/dist/index.d.ts +23 -0
- package/dist/src/admin/cli-routes.d.ts +22 -0
- package/dist/src/admin/env-file.d.ts +1 -0
- package/dist/src/admin/model-routes.d.ts +26 -0
- package/dist/src/admin/qdrant-scroll.d.ts +34 -0
- package/dist/src/admin/server.d.ts +40 -0
- package/dist/src/admin/wiki-routes.d.ts +31 -0
- package/dist/src/compose.d.ts +42 -0
- package/dist/src/config/define-config.d.ts +31 -0
- package/dist/src/cron/idle-session-cron.d.ts +80 -0
- package/dist/src/cron/idle-session-scanner.d.ts +16 -0
- package/dist/src/cron/self-review-cron.d.ts +55 -0
- package/dist/src/cron/semantic-consolidation.d.ts +71 -0
- package/dist/src/memory/embedder.d.ts +9 -0
- package/dist/src/memory/episodic-store.d.ts +121 -0
- package/dist/src/memory/memory-provider.d.ts +51 -0
- package/dist/src/memory/semantic-facts-store.d.ts +37 -0
- package/dist/src/memory/tool-corrections-store.d.ts +26 -0
- package/dist/src/memory/verbatim-archive-store.d.ts +86 -0
- package/dist/src/model/client.d.ts +24 -0
- package/dist/src/model/context-size.d.ts +30 -0
- package/dist/src/plugins/manifest.d.ts +29 -0
- package/dist/src/plugins/plugin-loader.d.ts +85 -0
- package/dist/src/router/channel-loader.d.ts +30 -0
- package/dist/src/router/provider.d.ts +7 -0
- package/dist/src/router/terminal-provider.d.ts +37 -0
- package/dist/src/router/terminal.d.ts +41 -0
- package/dist/src/router/tool-log.d.ts +65 -0
- package/dist/src/router/turn-runner.d.ts +86 -0
- package/dist/src/session/agent-turn.d.ts +266 -0
- package/dist/src/session/context-primer.d.ts +16 -0
- package/dist/src/session/episodic-summarizer.d.ts +25 -0
- package/dist/src/session/history.d.ts +95 -0
- package/dist/src/session/pending-confirmation.d.ts +8 -0
- package/dist/src/session/read-skill-tool.d.ts +4 -0
- package/dist/src/session/semantic-fact-extractor.d.ts +45 -0
- package/dist/src/session/step-info.d.ts +24 -0
- package/dist/src/session/summarizer.d.ts +23 -0
- package/dist/src/session/system-prompt.d.ts +38 -0
- package/dist/src/session/tool-correction-extractor.d.ts +43 -0
- package/dist/src/session/tool-log-buffer.d.ts +24 -0
- package/dist/src/session/tool-log-recall-tool.d.ts +18 -0
- package/dist/src/session/tool-start-hook.d.ts +57 -0
- package/dist/src/tools/display-store.d.ts +36 -0
- package/dist/src/tools/present-tool.d.ts +23 -0
- package/dist/src/wiki/frontmatter-schema.d.ts +53 -0
- package/dist/src/wiki/index-entry.d.ts +15 -0
- package/dist/src/wiki/orphan-detector.d.ts +1 -0
- package/dist/src/wiki/self-review-runner.d.ts +48 -0
- package/dist/src/wiki/self-review-tools.d.ts +22 -0
- package/dist/src/wiki/vault-cli.d.ts +2 -0
- package/dist/src/wiki/vault-init.d.ts +7 -0
- package/dist/src/wiki/wiki-note.d.ts +62 -0
- package/dist/src/wiki/wiki-read.d.ts +27 -0
- package/dist/src/wiki/wiki-tools.d.ts +7 -0
- package/index.ts +23 -0
- package/package.json +49 -0
- package/src/admin/cli-routes.ts +48 -0
- package/src/admin/env-file.ts +29 -0
- package/src/admin/model-routes.ts +71 -0
- package/src/admin/public/index.html +416 -0
- package/src/admin/qdrant-scroll.ts +45 -0
- package/src/admin/server.ts +188 -0
- package/src/admin/wiki-routes.ts +93 -0
- package/src/compose.ts +599 -0
- package/src/config/define-config.ts +35 -0
- package/src/cron/.gitkeep +0 -0
- package/src/cron/idle-session-cron.ts +144 -0
- package/src/cron/idle-session-scanner.ts +37 -0
- package/src/cron/self-review-cron.ts +103 -0
- package/src/cron/semantic-consolidation.ts +228 -0
- package/src/memory/.gitkeep +0 -0
- package/src/memory/embedder.ts +15 -0
- package/src/memory/episodic-store.ts +183 -0
- package/src/memory/memory-provider.ts +98 -0
- package/src/memory/semantic-facts-store.ts +89 -0
- package/src/memory/tool-corrections-store.ts +72 -0
- package/src/memory/verbatim-archive-store.ts +202 -0
- package/src/model/client.ts +33 -0
- package/src/model/context-size.ts +42 -0
- package/src/plugins/manifest.ts +47 -0
- package/src/plugins/plugin-loader.ts +205 -0
- package/src/router/channel-loader.ts +56 -0
- package/src/router/provider.ts +7 -0
- package/src/router/terminal-provider.ts +155 -0
- package/src/router/terminal.ts +151 -0
- package/src/router/tool-log.ts +116 -0
- package/src/router/turn-runner.ts +205 -0
- package/src/session/agent-turn.ts +391 -0
- package/src/session/context-primer.ts +134 -0
- package/src/session/episodic-summarizer.ts +38 -0
- package/src/session/history.ts +168 -0
- package/src/session/pending-confirmation.ts +8 -0
- package/src/session/read-skill-tool.ts +38 -0
- package/src/session/semantic-fact-extractor.ts +69 -0
- package/src/session/step-info.ts +27 -0
- package/src/session/summarizer.ts +36 -0
- package/src/session/system-prompt.ts +142 -0
- package/src/session/tool-correction-extractor.ts +133 -0
- package/src/session/tool-log-buffer.ts +73 -0
- package/src/session/tool-log-recall-tool.ts +38 -0
- package/src/session/tool-start-hook.ts +164 -0
- package/src/tools/display-store.ts +89 -0
- package/src/tools/present-tool.ts +41 -0
- package/src/wiki/.gitkeep +0 -0
- package/src/wiki/frontmatter-schema.ts +49 -0
- package/src/wiki/index-entry.ts +59 -0
- package/src/wiki/orphan-detector.ts +61 -0
- package/src/wiki/self-review-runner.ts +133 -0
- package/src/wiki/self-review-tools.ts +162 -0
- package/src/wiki/vault-cli.ts +143 -0
- package/src/wiki/vault-init.ts +43 -0
- package/src/wiki/wiki-note.ts +326 -0
- package/src/wiki/wiki-read.ts +122 -0
- package/src/wiki/wiki-tools.ts +112 -0
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Orchestrates a single conversational turn: append the user's input to
|
|
3
|
+
* history, ask the model for a response (with access to whichever tools
|
|
4
|
+
* this Mercury instance has wired in), append the response to history,
|
|
5
|
+
* return it.
|
|
6
|
+
*
|
|
7
|
+
* This is the one place that calls `generateText` for the main agent
|
|
8
|
+
* loop (as opposed to `src/session/summarizer.ts`, which calls it for
|
|
9
|
+
* summarization) — see `buildGenerateTextParams` below for why it's
|
|
10
|
+
* `ai-sdk-ollama`'s enhanced version, not the plain `ai` one. The real
|
|
11
|
+
* call is injected as
|
|
12
|
+
* `generateTextFn` so tests can exercise the sequencing/wiring logic
|
|
13
|
+
* here without needing a real model — see the test file for what that
|
|
14
|
+
* does and doesn't cover.
|
|
15
|
+
*
|
|
16
|
+
* Deliberately generic about *what* Mercury can do: `system` and
|
|
17
|
+
* `tools` are both passed in by the caller rather than hardcoded here.
|
|
18
|
+
* A fixed prompt baked into this file describing a specific tool (e.g.
|
|
19
|
+
* "use jiraCli") would be actively wrong on an instance where that tool
|
|
20
|
+
* isn't wired in — the model could still attempt a call to a tool name
|
|
21
|
+
* not present in the request's tool schema, which the AI SDK surfaces
|
|
22
|
+
* as a real error (`NoSuchToolError`), not a harmless no-op. Composing
|
|
23
|
+
* a `system` string that accurately reflects which tools are actually
|
|
24
|
+
* available is `src/index.ts`'s job, since that's where the set of
|
|
25
|
+
* enabled CLIs/tools is decided.
|
|
26
|
+
*
|
|
27
|
+
* Used by: `src/router/turn-runner.ts`'s `createTurnRunner`, the one
|
|
28
|
+
* shared driver every `Provider` (terminal, Google Chat) funnels its
|
|
29
|
+
* turns through (see `src/router/provider.ts`) — `runTurn` itself doesn't
|
|
30
|
+
* know or care which provider a given conversation came from.
|
|
31
|
+
*
|
|
32
|
+
* Streaming (see `buildStreamTextParams`/`runTurn`'s `onTextChunk`) is
|
|
33
|
+
* opt-in: providing `onTextChunk` *or* `onReasoningChunk` switches this
|
|
34
|
+
* call to `streamText`, reading its `fullStream` instead of waiting in
|
|
35
|
+
* silence for a full response — which can take several seconds on the
|
|
36
|
+
* local development model. The terminal provider supplies both (prints
|
|
37
|
+
* the answer and the model's live reasoning to stdout); Google Chat's
|
|
38
|
+
* `TurnSink` supplies only `onReasoningChunk` (a live-patched status card
|
|
39
|
+
* for the model's reasoning) and deliberately never `onTextChunk` — Chat
|
|
40
|
+
* only shows a message once it's fully sent, so incremental *answer*
|
|
41
|
+
* delivery never actually reaches a human faster there, unlike a
|
|
42
|
+
* terminal's live-updating stdout; the reasoning card is a different,
|
|
43
|
+
* already-patchable surface (see `google-chat-provider.ts`'s `createSink`).
|
|
44
|
+
* Either way this function still returns the full answer text and records
|
|
45
|
+
* it as one assistant history entry, same as the plain `generateText` path
|
|
46
|
+
* below — reasoning content is never part of `fullText` or the recorded
|
|
47
|
+
* history entry, it only ever reaches `onReasoningChunk`/`onReasoningEnd`,
|
|
48
|
+
* since it's a live UI-only surface, not something the model should ever
|
|
49
|
+
* see reflected back at it on a later turn.
|
|
50
|
+
*/
|
|
51
|
+
import { isStepCount, type LanguageModel, type StopCondition, type Tool } from "ai";
|
|
52
|
+
import { generateText, streamText } from "ai-sdk-ollama";
|
|
53
|
+
import type { Message, SessionHistory } from "./history.ts";
|
|
54
|
+
import type { StepInfo } from "./step-info.ts";
|
|
55
|
+
import { detectPendingConfirmation } from "./pending-confirmation.ts";
|
|
56
|
+
// The sentinel is a channel-side contract value (see `@mercury-fw/channel-types`);
|
|
57
|
+
// re-exported here for the core callers importing it from this module.
|
|
58
|
+
import { PENDING_CONFIRMATION_NOTE } from "@mercury-fw/channel-types";
|
|
59
|
+
export { PENDING_CONFIRMATION_NOTE };
|
|
60
|
+
|
|
61
|
+
/** `""` unless `lastStep` staged a confirm-required command, in which case `PENDING_CONFIRMATION_NOTE`. */
|
|
62
|
+
function resolveEmptyText(lastStep: StepInfo | undefined): string {
|
|
63
|
+
return lastStep && detectPendingConfirmation(lastStep) ? PENDING_CONFIRMATION_NOTE : "";
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* What actually gets recorded into `SessionHistory` for this turn — `displayText`
|
|
68
|
+
* unchanged, except when `lastStep` staged a confirm-required command: then
|
|
69
|
+
* it's the opaque `[REQ:<token>]` marker instead, regardless of `displayText`.
|
|
70
|
+
* The two diverge on purpose: `displayText` (what the channel shows/streams)
|
|
71
|
+
* stays the fixed `PENDING_CONFIRMATION_NOTE` sentence — both channels
|
|
72
|
+
* already suppress that exact string and show their own UI instead (a card,
|
|
73
|
+
* a printed instruction) — but that sentence must never reach persistent
|
|
74
|
+
* memory. A confirmation token is ephemeral by design (in-memory only, a
|
|
75
|
+
* few minutes' TTL — see `confirmation-store.ts`), so any summarized record
|
|
76
|
+
* claiming an action is "still pending" goes stale almost immediately;
|
|
77
|
+
* this is the stale-primer bug's actual fix. The marker forces a reader
|
|
78
|
+
* (the model, later, via `context-primer.ts`'s "Riferimenti aperti" +
|
|
79
|
+
* `resolve_reference` in `wiki-tools.ts`) to deliberately look up what it
|
|
80
|
+
* refers to instead of absorbing a claim that may no longer be true.
|
|
81
|
+
*/
|
|
82
|
+
function resolveHistoryText(lastStep: StepInfo | undefined, displayText: string): string {
|
|
83
|
+
const pending = lastStep ? detectPendingConfirmation(lastStep) : null;
|
|
84
|
+
return pending ? `[REQ:${pending.token}]` : displayText;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Stops the tool-calling loop immediately after a step that staged a
|
|
89
|
+
* confirm-required command — the model never gets a further step to
|
|
90
|
+
* write anything about it (see `PENDING_CONFIRMATION_NOTE` above for
|
|
91
|
+
* why). `stopWhen` conditions are OR'd together (confirmed against the
|
|
92
|
+
* AI SDK's own source: `isStopConditionMet` resolves every condition and
|
|
93
|
+
* stops if *any* is true), so combining this with `stepCountIs(100)`
|
|
94
|
+
* doesn't change when the step-budget cap itself kicks in.
|
|
95
|
+
*/
|
|
96
|
+
function pendingConfirmationStop(): StopCondition<any> {
|
|
97
|
+
return ({ steps }) => {
|
|
98
|
+
const last = steps[steps.length - 1];
|
|
99
|
+
return last !== undefined && detectPendingConfirmation(last) !== null;
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** The shape of the AI SDK call this module needs, injectable for tests. */
|
|
104
|
+
type GenerateTextFn = (params: {
|
|
105
|
+
model: LanguageModel;
|
|
106
|
+
messages: Message[];
|
|
107
|
+
tools: Record<string, Tool>;
|
|
108
|
+
instructions: string;
|
|
109
|
+
onStepEnd?: (step: StepInfo) => void;
|
|
110
|
+
/** Aborts the in-flight generation when the caller cancels the turn. */
|
|
111
|
+
abortSignal?: AbortSignal;
|
|
112
|
+
}) => Promise<{ text: string; usage?: { inputTokens: number | undefined } }>;
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Minimal shape this file reads off a `fullStream` part — deliberately a
|
|
116
|
+
* loose supertype of the real (much larger) `TextStreamPart` union `ai`
|
|
117
|
+
* actually emits (`node_modules/ai/dist/index.d.ts:2484`: `start`,
|
|
118
|
+
* `text-start`, `tool-call`, `finish`, etc.), not a closed enumeration of
|
|
119
|
+
* it: a closed union here would reject every part type this file doesn't
|
|
120
|
+
* care about, which the real stream emits plenty of. `delta`/`text` are
|
|
121
|
+
* both optional for the same reason — only present on the part types this
|
|
122
|
+
* file actually reads (`text-delta`/`reasoning-delta`), and the *name* of
|
|
123
|
+
* that field is itself unstable: `ai`'s own `.d.ts` declares this part
|
|
124
|
+
* shape twice with different field names for the same `type` value
|
|
125
|
+
* (`node_modules/ai/dist/index.d.ts:2103-2107` uses `delta`, `2555-2558`
|
|
126
|
+
* uses `text`) — confirmed live against the real `ai-sdk-ollama` stream
|
|
127
|
+
* that the installed version actually emits `text`, not `delta`. Reading
|
|
128
|
+
* both defensively (see the loop below) survives either shape rather than
|
|
129
|
+
* silently reading `undefined` and going empty if a future upgrade flips
|
|
130
|
+
* it back.
|
|
131
|
+
*/
|
|
132
|
+
type StreamPart = { type: string; delta?: string; text?: string; id?: string };
|
|
133
|
+
|
|
134
|
+
/** The shape of the AI SDK streaming call this module needs, injectable for tests. */
|
|
135
|
+
type StreamTextFn = (params: {
|
|
136
|
+
model: LanguageModel;
|
|
137
|
+
messages: Message[];
|
|
138
|
+
tools: Record<string, Tool>;
|
|
139
|
+
instructions: string;
|
|
140
|
+
onStepEnd?: (step: StepInfo) => void;
|
|
141
|
+
/** Aborts the in-flight generation when the caller cancels the turn. */
|
|
142
|
+
abortSignal?: AbortSignal;
|
|
143
|
+
}) => Promise<{
|
|
144
|
+
stream: AsyncIterable<StreamPart>;
|
|
145
|
+
usage?: PromiseLike<{ inputTokens: number | undefined }>;
|
|
146
|
+
}>;
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Builds the params object passed to the real `generateText`. Extracted
|
|
150
|
+
* as its own pure function so it's unit-testable without a real model —
|
|
151
|
+
* `defaultGenerateTextFn` below is otherwise a thin, untestable wrapper
|
|
152
|
+
* around a direct SDK call.
|
|
153
|
+
*
|
|
154
|
+
* `stopWhen: stepCountIs(100)` is the one non-obvious part: `generateText`
|
|
155
|
+
* defaults to stopping after a single step. If that step is a tool call
|
|
156
|
+
* with no accompanying text (the normal case — the model calls a tool,
|
|
157
|
+
* then needs the tool's result before it can answer), the call returns
|
|
158
|
+
* with `text: ""` and no error, having never given the model a chance to
|
|
159
|
+
* read the tool result and respond — some cap above 1 is needed for
|
|
160
|
+
* that. Raised from an original 5 (observed live: a conversation
|
|
161
|
+
* combining wiki lookups with a jira search — which always needs
|
|
162
|
+
* `--select`, so almost always costs 2 attempts — routinely burned all 5
|
|
163
|
+
* steps on tool calls alone, leaving the model zero steps to ever write
|
|
164
|
+
* an answer) to 20 (matching this SDK's own default for its
|
|
165
|
+
* higher-level agent construct, `ToolLoopAgentSettings`), then to 100 for
|
|
166
|
+
* extra headroom during this research phase — the model should be free
|
|
167
|
+
* to make as many tool calls as it actually needs, not be cut off by an
|
|
168
|
+
* arbitrary ceiling. `runTurn`'s empty-text fallback (below) is the
|
|
169
|
+
* remaining safety net regardless of the cap's value: if a turn still
|
|
170
|
+
* exhausts it with nothing to show, that's communicated explicitly
|
|
171
|
+
* rather than left silent — the real backstop against a runaway loop is
|
|
172
|
+
* that fallback plus the step count itself still being finite, not the
|
|
173
|
+
* specific number.
|
|
174
|
+
*
|
|
175
|
+
* `generateText` itself is imported from `ai-sdk-ollama`, not the plain
|
|
176
|
+
* `ai` package — confirmed by a real run: with the standard SDK's
|
|
177
|
+
* version, Ollama executed the tool call but `text` still came back
|
|
178
|
+
* empty even with multi-step enabled. `ai-sdk-ollama`'s enhanced
|
|
179
|
+
* `generateText` is a drop-in replacement (same params/return shape)
|
|
180
|
+
* that specifically synthesizes a real response when this happens —
|
|
181
|
+
* documented as a known Ollama-provider quirk, not something to patch
|
|
182
|
+
* around here by hand.
|
|
183
|
+
*/
|
|
184
|
+
export function buildGenerateTextParams(params: {
|
|
185
|
+
model: LanguageModel;
|
|
186
|
+
messages: Message[];
|
|
187
|
+
tools: Record<string, Tool>;
|
|
188
|
+
instructions: string;
|
|
189
|
+
onStepEnd?: (step: StepInfo) => void;
|
|
190
|
+
abortSignal?: AbortSignal;
|
|
191
|
+
}) {
|
|
192
|
+
return { ...params, stopWhen: [isStepCount(100), pendingConfirmationStop()] };
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** Default production implementation: calls the real `generateText` from `ai`. */
|
|
196
|
+
const defaultGenerateTextFn: GenerateTextFn = (params) =>
|
|
197
|
+
generateText(buildGenerateTextParams(params));
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Builds the params object passed to the real `streamText`. Same
|
|
201
|
+
* `stopWhen` reasoning as `buildGenerateTextParams` — a tool-call-only
|
|
202
|
+
* first step must not be the stream's last step either. `ai-sdk-ollama`'s
|
|
203
|
+
* `streamText` carries its own equivalent of the empty-text-after-
|
|
204
|
+
* tool-call fix (`enableStreamingSynthesis`, on by default) — verified by
|
|
205
|
+
* reading its source before relying on it, not assumed from the
|
|
206
|
+
* non-streaming behavior.
|
|
207
|
+
*/
|
|
208
|
+
export function buildStreamTextParams(params: {
|
|
209
|
+
model: LanguageModel;
|
|
210
|
+
messages: Message[];
|
|
211
|
+
tools: Record<string, Tool>;
|
|
212
|
+
instructions: string;
|
|
213
|
+
onStepEnd?: (step: StepInfo) => void;
|
|
214
|
+
abortSignal?: AbortSignal;
|
|
215
|
+
}) {
|
|
216
|
+
return { ...params, stopWhen: [isStepCount(100), pendingConfirmationStop()] };
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/** Default production implementation: calls the real `streamText` from `ai-sdk-ollama`. */
|
|
220
|
+
const defaultStreamTextFn: StreamTextFn = (params) => streamText(buildStreamTextParams(params));
|
|
221
|
+
|
|
222
|
+
/**
|
|
223
|
+
* Runs one turn: records `userInput`, generates a response with the
|
|
224
|
+
* given model/tools/system prompt, records the response, and returns it.
|
|
225
|
+
*
|
|
226
|
+
* @param history - The conversation's `SessionHistory`; one per
|
|
227
|
+
* channel/space, never shared (see `src/index.ts`).
|
|
228
|
+
* @param userInput - The user's message for this turn.
|
|
229
|
+
* @param deps.model - The language model to use.
|
|
230
|
+
* @param deps.tools - The tools available to the model on this call —
|
|
231
|
+
* which tools end up here depends on which CLIs this Mercury instance
|
|
232
|
+
* has enabled (see `src/index.ts`), not on anything in this file.
|
|
233
|
+
* @param deps.system - The system prompt for this call. Must accurately
|
|
234
|
+
* describe only the tools actually present in `deps.tools` — this
|
|
235
|
+
* function doesn't validate that, the caller is responsible for
|
|
236
|
+
* keeping the two in sync.
|
|
237
|
+
* @param deps.onStepFinish - Optional, called once per generation step
|
|
238
|
+
* (including intermediate ones with only a tool call, no text).
|
|
239
|
+
* `src/router/turn-runner.ts`'s shared `createTurnRunner` wires this for
|
|
240
|
+
* every provider (fans out to `logStep`/`recordStep`/the provider's own
|
|
241
|
+
* `TurnSink.onStep`) — what each provider actually *does* with a step
|
|
242
|
+
* still differs (the terminal prints it, Google Chat's `TurnSink`
|
|
243
|
+
* doesn't define `onStep` at all, showing raw tool calls to a chat
|
|
244
|
+
* audience isn't the same call as showing them to whoever's debugging at
|
|
245
|
+
* a terminal), but the wiring itself is no longer channel-specific.
|
|
246
|
+
* @param deps.onTextChunk - Optional. When provided (alongside or instead
|
|
247
|
+
* of `onReasoningChunk`), this turn uses `streamText` instead of
|
|
248
|
+
* `generateText`, calling this once per answer-text chunk as it arrives
|
|
249
|
+
* — see the file header for why Google Chat never sets this one. The
|
|
250
|
+
* returned string and the recorded history entry are the same either
|
|
251
|
+
* way: the full answer text, joined from every `text-delta` chunk only.
|
|
252
|
+
* @param deps.onReasoningChunk - Optional. Also switches this turn onto
|
|
253
|
+
* `streamText` (see `onTextChunk`). Called once per reasoning-token
|
|
254
|
+
* delta as it streams, tagged with the SDK's own id for that reasoning
|
|
255
|
+
* block — only ever fires for a model that actually supports Ollama's
|
|
256
|
+
* native extended thinking (see `src/index.ts`'s `OLLAMA_THINK`);
|
|
257
|
+
* otherwise no reasoning parts ever arrive and this is simply never
|
|
258
|
+
* called. This content is UI-only: it never touches the returned text
|
|
259
|
+
* or `SessionHistory`. A single turn can reason more than once (e.g.
|
|
260
|
+
* once before a tool call, again after seeing its result) — each burst
|
|
261
|
+
* carries its own id, letting a caller (e.g. a status card per id)
|
|
262
|
+
* treat them as independent rather than one continuous stream.
|
|
263
|
+
* @param deps.onReasoningEnd - Optional. Fires once per reasoning block
|
|
264
|
+
* that actually started (i.e. once per distinct id `onReasoningChunk`
|
|
265
|
+
* reported) — including if the stream aborts while a block is still
|
|
266
|
+
* open, so a caller building a live display (a status card, a printed
|
|
267
|
+
* block) can't be left stuck open forever. `failed` is `true` only for
|
|
268
|
+
* that abrupt-abort case, `false` on a normal reasoning-end.
|
|
269
|
+
* @param deps.generateTextFn - Test seam for the non-streaming path;
|
|
270
|
+
* defaults to the real AI SDK call. Injecting a fake here only tests
|
|
271
|
+
* this function's own sequencing — it does not exercise the real model
|
|
272
|
+
* or the real AI SDK integration, which can only be verified by an
|
|
273
|
+
* actual end-to-end run.
|
|
274
|
+
* @param deps.streamTextFn - Test seam for the streaming path (used when
|
|
275
|
+
* `onTextChunk`/`onReasoningChunk` is provided), same caveat as
|
|
276
|
+
* `generateTextFn`.
|
|
277
|
+
* @param deps.onUsage - Optional, called once per turn with the real
|
|
278
|
+
* `inputTokens` count the model actually consumed (summed across every
|
|
279
|
+
* step, including tool calls) — not an estimate. The terminal channel
|
|
280
|
+
* uses this for a real context-usage indicator (see
|
|
281
|
+
* `src/router/tool-log.ts`'s `formatContextUsage`); Mercury's own char-
|
|
282
|
+
* count heuristic in `src/session/history.ts` is for a different
|
|
283
|
+
* purpose (deciding when to summarize) and intentionally untouched by
|
|
284
|
+
* this.
|
|
285
|
+
*/
|
|
286
|
+
export async function runTurn(
|
|
287
|
+
history: SessionHistory,
|
|
288
|
+
userInput: string,
|
|
289
|
+
deps: {
|
|
290
|
+
model: LanguageModel;
|
|
291
|
+
tools: Record<string, Tool>;
|
|
292
|
+
system: string;
|
|
293
|
+
onStepFinish?: (step: StepInfo) => void;
|
|
294
|
+
onTextChunk?: (chunk: string) => void;
|
|
295
|
+
onReasoningChunk?: (chunk: string, id: string) => void;
|
|
296
|
+
onReasoningEnd?: (id: string, failed: boolean) => void;
|
|
297
|
+
onUsage?: (inputTokens: number | undefined) => void;
|
|
298
|
+
generateTextFn?: GenerateTextFn;
|
|
299
|
+
streamTextFn?: StreamTextFn;
|
|
300
|
+
/** When set, aborting it stops the in-flight generation — the turn ends
|
|
301
|
+
* promptly instead of running to completion (see the HTTP surface's
|
|
302
|
+
* client-disconnect cancellation). Optional; existing callers pass none. */
|
|
303
|
+
abortSignal?: AbortSignal;
|
|
304
|
+
},
|
|
305
|
+
): Promise<string> {
|
|
306
|
+
await history.addUserMessage(userInput);
|
|
307
|
+
|
|
308
|
+
if (deps.onTextChunk || deps.onReasoningChunk) {
|
|
309
|
+
const stream = deps.streamTextFn ?? defaultStreamTextFn;
|
|
310
|
+
let lastStep: StepInfo | undefined;
|
|
311
|
+
const result = await stream({
|
|
312
|
+
model: deps.model,
|
|
313
|
+
messages: history.getMessages(),
|
|
314
|
+
tools: deps.tools,
|
|
315
|
+
instructions: deps.system,
|
|
316
|
+
abortSignal: deps.abortSignal,
|
|
317
|
+
onStepEnd: (step) => {
|
|
318
|
+
lastStep = step;
|
|
319
|
+
deps.onStepFinish?.(step);
|
|
320
|
+
},
|
|
321
|
+
});
|
|
322
|
+
|
|
323
|
+
let fullText = "";
|
|
324
|
+
// The id of the reasoning block currently open, if any — a tool-
|
|
325
|
+
// calling turn can reason multiple times (before a tool call, again
|
|
326
|
+
// after seeing its result, ...), each burst carrying its own id from
|
|
327
|
+
// the SDK. Tracking only the *current* one (not a set of all-ever-seen
|
|
328
|
+
// ids) is enough: it's `undefined` whenever no block is open, which is
|
|
329
|
+
// exactly when the abrupt-failure guard below must stay silent.
|
|
330
|
+
let currentReasoningId: string | undefined;
|
|
331
|
+
try {
|
|
332
|
+
for await (const part of result.stream) {
|
|
333
|
+
if (part.type === "text-delta") {
|
|
334
|
+
const delta = part.text ?? part.delta ?? "";
|
|
335
|
+
fullText += delta;
|
|
336
|
+
deps.onTextChunk?.(delta);
|
|
337
|
+
} else if (part.type === "reasoning-delta") {
|
|
338
|
+
const id = part.id ?? "";
|
|
339
|
+
currentReasoningId = id;
|
|
340
|
+
deps.onReasoningChunk?.(part.text ?? part.delta ?? "", id);
|
|
341
|
+
} else if (part.type === "reasoning-end") {
|
|
342
|
+
const id = part.id ?? "";
|
|
343
|
+
currentReasoningId = undefined;
|
|
344
|
+
deps.onReasoningEnd?.(id, false);
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
} finally {
|
|
348
|
+
// Guarantees onReasoningEnd fires even if the stream throws
|
|
349
|
+
// mid-reasoning (no reasoning-end part arrives on an abrupt
|
|
350
|
+
// failure) — a caller building a live display (a status card, a
|
|
351
|
+
// printed block) can't be left stuck open forever. Only fires for
|
|
352
|
+
// whichever block was actually open when the failure happened
|
|
353
|
+
// (`currentReasoningId` is cleared by a normal reasoning-end, so
|
|
354
|
+
// this is a no-op outside an open block). `failed: true` here (vs.
|
|
355
|
+
// `false` on a normal reasoning-end above) is what lets a caller
|
|
356
|
+
// like Google Chat's status card show the right outcome — without
|
|
357
|
+
// it, every abrupt failure would look identical to a clean finish.
|
|
358
|
+
if (currentReasoningId !== undefined) {
|
|
359
|
+
deps.onReasoningEnd?.(currentReasoningId, true);
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
deps.onUsage?.((await result.usage)?.inputTokens);
|
|
363
|
+
|
|
364
|
+
if (fullText.trim().length === 0) {
|
|
365
|
+
fullText = resolveEmptyText(lastStep);
|
|
366
|
+
if (fullText.length > 0) deps.onTextChunk?.(fullText);
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
await history.addAssistantMessage(resolveHistoryText(lastStep, fullText));
|
|
370
|
+
return fullText;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
const generate = deps.generateTextFn ?? defaultGenerateTextFn;
|
|
374
|
+
let lastStep: StepInfo | undefined;
|
|
375
|
+
const result = await generate({
|
|
376
|
+
model: deps.model,
|
|
377
|
+
messages: history.getMessages(),
|
|
378
|
+
tools: deps.tools,
|
|
379
|
+
instructions: deps.system,
|
|
380
|
+
abortSignal: deps.abortSignal,
|
|
381
|
+
onStepEnd: (step) => {
|
|
382
|
+
lastStep = step;
|
|
383
|
+
deps.onStepFinish?.(step);
|
|
384
|
+
},
|
|
385
|
+
});
|
|
386
|
+
deps.onUsage?.(result.usage?.inputTokens);
|
|
387
|
+
|
|
388
|
+
const text = result.text.trim().length === 0 ? resolveEmptyText(lastStep) : result.text;
|
|
389
|
+
await history.addAssistantMessage(resolveHistoryText(lastStep, text));
|
|
390
|
+
return text;
|
|
391
|
+
}
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Builds the synthetic primer text seeded into a brand-new Google Chat
|
|
3
|
+
* session's history (see `history.ts`'s `primer` param): the wiki's own
|
|
4
|
+
* index (same for every user, not tied to any prior session), the wiki
|
|
5
|
+
* facts the user's last closed session actually reinforced, plus that same
|
|
6
|
+
* session's own episodic entries. Empty string when there's nothing —
|
|
7
|
+
* enrichment, a caller must work fine without it. Deliberately not
|
|
8
|
+
* similarity retrieval: there's no query yet to compare against at session
|
|
9
|
+
* start, only "what happened last time" (see `getLastSessionEpisodicSummaries`)
|
|
10
|
+
* plus "what's in the wiki right now".
|
|
11
|
+
*/
|
|
12
|
+
import { resolve } from "node:path";
|
|
13
|
+
import { parse as parseYaml } from "yaml";
|
|
14
|
+
import type { EpisodicSummary } from "../memory/episodic-store.ts";
|
|
15
|
+
import type { listWikiFilesInRoots, readWikiFileInRoots, readIndexFile } from "../wiki/wiki-read.ts";
|
|
16
|
+
|
|
17
|
+
export type ContextPrimerDeps = {
|
|
18
|
+
vaultPath: string;
|
|
19
|
+
/** Already scoped to the last closed session's own sessionKey — see `getLastSessionEpisodicSummaries`. */
|
|
20
|
+
getLastSessionEntries: (userId: string) => Promise<EpisodicSummary[]>;
|
|
21
|
+
listWikiFilesInRootsFn: typeof listWikiFilesInRoots;
|
|
22
|
+
readWikiFileInRootsFn: typeof readWikiFileInRoots;
|
|
23
|
+
readIndexFileFn: typeof readIndexFile;
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
const FRONTMATTER_RE = /^---\n([\s\S]*?)\n---\n/;
|
|
27
|
+
|
|
28
|
+
/** Same frontmatter shape `semantic-consolidation.ts` already parses (`derived_from: string[]`) — only `inferred` notes carry it, a `resolved`/`curated` note yields none. */
|
|
29
|
+
function derivedFromTimestamps(text: string): string[] {
|
|
30
|
+
const match = FRONTMATTER_RE.exec(text);
|
|
31
|
+
if (!match) {
|
|
32
|
+
return [];
|
|
33
|
+
}
|
|
34
|
+
const frontmatter = parseYaml(match[1] as string) as { derived_from?: unknown };
|
|
35
|
+
return Array.isArray(frontmatter.derived_from) ? frontmatter.derived_from : [];
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** The note's content after its frontmatter block — matches `writeNoteFile`'s `---\n<yaml>---\n\n<body>\n` layout. */
|
|
39
|
+
function noteBody(text: string): string {
|
|
40
|
+
const match = FRONTMATTER_RE.exec(text);
|
|
41
|
+
return (match ? text.slice(match[0].length) : text).trim();
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** Extracts a confirmation note's `status` field — `undefined` for anything not shaped like one (missing/unparseable frontmatter). */
|
|
45
|
+
function noteStatus(text: string): string | undefined {
|
|
46
|
+
const match = FRONTMATTER_RE.exec(text);
|
|
47
|
+
if (!match) {
|
|
48
|
+
return undefined;
|
|
49
|
+
}
|
|
50
|
+
const frontmatter = parseYaml(match[1] as string) as { status?: unknown };
|
|
51
|
+
return typeof frontmatter.status === "string" ? frontmatter.status : undefined;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Tokens of this user's still-`"pending"` confirmation notes
|
|
56
|
+
* (`inferred/confirmations/<userId>/<token>.md` — see `writeConfirmationNote`
|
|
57
|
+
* in `wiki-note.ts`). Deliberately a separate lookup from the "Known facts"
|
|
58
|
+
* cross-reference below: this subtree isn't correlated to the last
|
|
59
|
+
* session's episodic timestamps, it's simply "whatever is still open right
|
|
60
|
+
* now" for this user, regardless of which session staged it.
|
|
61
|
+
*/
|
|
62
|
+
async function pendingConfirmationTokens(userId: string, deps: ContextPrimerDeps): Promise<string[]> {
|
|
63
|
+
const confirmationsRoot = resolve(deps.vaultPath, "inferred", "confirmations", encodeURIComponent(userId));
|
|
64
|
+
const files = await deps.listWikiFilesInRootsFn(deps.vaultPath, [confirmationsRoot]);
|
|
65
|
+
const tokens: string[] = [];
|
|
66
|
+
for (const file of files) {
|
|
67
|
+
const text = await deps.readWikiFileInRootsFn(deps.vaultPath, [confirmationsRoot], file);
|
|
68
|
+
if (noteStatus(text) === "pending") {
|
|
69
|
+
tokens.push(file.split("/").pop()!.replace(/\.md$/, ""));
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
return tokens;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Text of the primer for `userId`, built from injected deps only — never
|
|
77
|
+
* touches Qdrant or the filesystem directly, so tests supply fakes and
|
|
78
|
+
* `index.ts` supplies the real Qdrant-backed episodic query and wiki reads.
|
|
79
|
+
*/
|
|
80
|
+
export async function buildContextPrimer(userId: string, deps: ContextPrimerDeps): Promise<string> {
|
|
81
|
+
const entries = await deps.getLastSessionEntries(userId);
|
|
82
|
+
// Checked regardless of `entries` — a pending confirmation isn't tied to
|
|
83
|
+
// "was there a prior episodic session", it's simply still open right now.
|
|
84
|
+
const pendingTokens = await pendingConfirmationTokens(userId, deps);
|
|
85
|
+
// Same for every user (curated/ has no per-user scoping) — checked
|
|
86
|
+
// regardless of prior session too, so even a first-ever session gets
|
|
87
|
+
// pointed at what's in the wiki right now.
|
|
88
|
+
const indexContent = (await deps.readIndexFileFn(deps.vaultPath)).trim();
|
|
89
|
+
|
|
90
|
+
if (entries.length === 0 && pendingTokens.length === 0 && indexContent.length === 0) {
|
|
91
|
+
return "";
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
const sections: string[] = [];
|
|
95
|
+
|
|
96
|
+
if (indexContent.length > 0) {
|
|
97
|
+
sections.push(`Wiki index:\n${indexContent}`);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
if (entries.length > 0) {
|
|
101
|
+
const entryTimestamps = new Set(entries.map((e) => e.timestamp));
|
|
102
|
+
const inferredRoot = resolve(deps.vaultPath, "inferred", "users", encodeURIComponent(userId));
|
|
103
|
+
const files = await deps.listWikiFilesInRootsFn(deps.vaultPath, [inferredRoot]);
|
|
104
|
+
|
|
105
|
+
const facts: string[] = [];
|
|
106
|
+
for (const file of files) {
|
|
107
|
+
const text = await deps.readWikiFileInRootsFn(deps.vaultPath, [inferredRoot], file);
|
|
108
|
+
const timestamps = derivedFromTimestamps(text);
|
|
109
|
+
if (!timestamps.some((ts) => entryTimestamps.has(ts))) {
|
|
110
|
+
continue;
|
|
111
|
+
}
|
|
112
|
+
const topic = file.split("/").pop()!.replace(/\.md$/, "");
|
|
113
|
+
const body = noteBody(text);
|
|
114
|
+
if (body) {
|
|
115
|
+
facts.push(`${topic}: ${body}`);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
if (facts.length > 0) {
|
|
120
|
+
sections.push(`Known facts:\n${facts.map((f) => `- ${f}`).join("\n")}`);
|
|
121
|
+
}
|
|
122
|
+
sections.push(`Last session:\n${entries.map((e) => `- ${e.summary}`).join("\n")}`);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// Deliberately opaque: just the token, no command, no "still pending"
|
|
126
|
+
// narrative — see pendingConfirmationTokens's own doc comment for why.
|
|
127
|
+
// Resolving it into something meaningful requires the model to
|
|
128
|
+
// deliberately call resolve_reference (wiki-tools.ts) with the token.
|
|
129
|
+
if (pendingTokens.length > 0) {
|
|
130
|
+
sections.push(`Riferimenti aperti:\n${pendingTokens.map((t) => `- [REQ:${t}]`).join("\n")}`);
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
return sections.join("\n\n");
|
|
134
|
+
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Thin glue turning a closed session's messages into the episodic
|
|
3
|
+
* summary written to Qdrant (see `src/memory/episodic-store.ts`) — a
|
|
4
|
+
* factual account of what happened, not an interpretation. Distinct from
|
|
5
|
+
* `summarizer.ts` (Layer 1): that one condenses history to keep the
|
|
6
|
+
* *next* turn's prompt small, preserving whatever helps continue the
|
|
7
|
+
* same conversation; this one produces a standalone record of a
|
|
8
|
+
* *finished* conversation, and must not infer patterns or preferences
|
|
9
|
+
* (that inference is a separate, deterministic consolidation step,
|
|
10
|
+
* not this LLM call's job).
|
|
11
|
+
*
|
|
12
|
+
* Deliberately never asks the model for a date: it has no reliable notion
|
|
13
|
+
* of "today" and would invent one (observed live: dates from the wrong
|
|
14
|
+
* year, hedged with "replace with current date if applicable"). The
|
|
15
|
+
* entry's own `timestamp` field (computed by the caller,
|
|
16
|
+
* `idle-session-cron.ts`) is the one and only source of truth for "when" —
|
|
17
|
+
* never duplicated into this text, mechanically or otherwise.
|
|
18
|
+
*
|
|
19
|
+
* Same "not worth mocking deeply" reasoning as `summarizer.ts` — no
|
|
20
|
+
* dedicated test file, it's one line of glue around `generateText`.
|
|
21
|
+
*/
|
|
22
|
+
import { generateText, type LanguageModel } from "ai";
|
|
23
|
+
import type { Message } from "./history.ts";
|
|
24
|
+
|
|
25
|
+
/** Returns a function that summarizes a closed session's messages into a factual account. */
|
|
26
|
+
export function createEpisodicSummarizer(
|
|
27
|
+
model: LanguageModel,
|
|
28
|
+
): (messages: Message[]) => Promise<string> {
|
|
29
|
+
return async (messages) => {
|
|
30
|
+
const { text } = await generateText({
|
|
31
|
+
model,
|
|
32
|
+
instructions:
|
|
33
|
+
"Summarize what happened in this conversation as a short, factual account for future recall — what was discussed, asked, or decided. Describe only what occurred in this session. Do not infer patterns, habits, or preferences about the user — that is a separate process, not this one. Never mention or guess a date — you don't reliably know the current date, and it's tracked separately from this text.",
|
|
34
|
+
prompt: messages.map((m) => `${m.role}: ${m.content}`).join("\n"),
|
|
35
|
+
});
|
|
36
|
+
return text;
|
|
37
|
+
};
|
|
38
|
+
}
|