@mercury-fw/core 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/README.md +38 -0
  3. package/dist/index.d.ts +23 -0
  4. package/dist/src/admin/cli-routes.d.ts +22 -0
  5. package/dist/src/admin/env-file.d.ts +1 -0
  6. package/dist/src/admin/model-routes.d.ts +26 -0
  7. package/dist/src/admin/qdrant-scroll.d.ts +34 -0
  8. package/dist/src/admin/server.d.ts +40 -0
  9. package/dist/src/admin/wiki-routes.d.ts +31 -0
  10. package/dist/src/compose.d.ts +42 -0
  11. package/dist/src/config/define-config.d.ts +31 -0
  12. package/dist/src/cron/idle-session-cron.d.ts +80 -0
  13. package/dist/src/cron/idle-session-scanner.d.ts +16 -0
  14. package/dist/src/cron/self-review-cron.d.ts +55 -0
  15. package/dist/src/cron/semantic-consolidation.d.ts +71 -0
  16. package/dist/src/memory/embedder.d.ts +9 -0
  17. package/dist/src/memory/episodic-store.d.ts +121 -0
  18. package/dist/src/memory/memory-provider.d.ts +51 -0
  19. package/dist/src/memory/semantic-facts-store.d.ts +37 -0
  20. package/dist/src/memory/tool-corrections-store.d.ts +26 -0
  21. package/dist/src/memory/verbatim-archive-store.d.ts +86 -0
  22. package/dist/src/model/client.d.ts +24 -0
  23. package/dist/src/model/context-size.d.ts +30 -0
  24. package/dist/src/plugins/manifest.d.ts +29 -0
  25. package/dist/src/plugins/plugin-loader.d.ts +85 -0
  26. package/dist/src/router/channel-loader.d.ts +30 -0
  27. package/dist/src/router/provider.d.ts +7 -0
  28. package/dist/src/router/terminal-provider.d.ts +37 -0
  29. package/dist/src/router/terminal.d.ts +41 -0
  30. package/dist/src/router/tool-log.d.ts +65 -0
  31. package/dist/src/router/turn-runner.d.ts +86 -0
  32. package/dist/src/session/agent-turn.d.ts +266 -0
  33. package/dist/src/session/context-primer.d.ts +16 -0
  34. package/dist/src/session/episodic-summarizer.d.ts +25 -0
  35. package/dist/src/session/history.d.ts +95 -0
  36. package/dist/src/session/pending-confirmation.d.ts +8 -0
  37. package/dist/src/session/read-skill-tool.d.ts +4 -0
  38. package/dist/src/session/semantic-fact-extractor.d.ts +45 -0
  39. package/dist/src/session/step-info.d.ts +24 -0
  40. package/dist/src/session/summarizer.d.ts +23 -0
  41. package/dist/src/session/system-prompt.d.ts +38 -0
  42. package/dist/src/session/tool-correction-extractor.d.ts +43 -0
  43. package/dist/src/session/tool-log-buffer.d.ts +24 -0
  44. package/dist/src/session/tool-log-recall-tool.d.ts +18 -0
  45. package/dist/src/session/tool-start-hook.d.ts +57 -0
  46. package/dist/src/tools/display-store.d.ts +36 -0
  47. package/dist/src/tools/present-tool.d.ts +23 -0
  48. package/dist/src/wiki/frontmatter-schema.d.ts +53 -0
  49. package/dist/src/wiki/index-entry.d.ts +15 -0
  50. package/dist/src/wiki/orphan-detector.d.ts +1 -0
  51. package/dist/src/wiki/self-review-runner.d.ts +48 -0
  52. package/dist/src/wiki/self-review-tools.d.ts +22 -0
  53. package/dist/src/wiki/vault-cli.d.ts +2 -0
  54. package/dist/src/wiki/vault-init.d.ts +7 -0
  55. package/dist/src/wiki/wiki-note.d.ts +62 -0
  56. package/dist/src/wiki/wiki-read.d.ts +27 -0
  57. package/dist/src/wiki/wiki-tools.d.ts +7 -0
  58. package/index.ts +23 -0
  59. package/package.json +49 -0
  60. package/src/admin/cli-routes.ts +48 -0
  61. package/src/admin/env-file.ts +29 -0
  62. package/src/admin/model-routes.ts +71 -0
  63. package/src/admin/public/index.html +416 -0
  64. package/src/admin/qdrant-scroll.ts +45 -0
  65. package/src/admin/server.ts +188 -0
  66. package/src/admin/wiki-routes.ts +93 -0
  67. package/src/compose.ts +599 -0
  68. package/src/config/define-config.ts +35 -0
  69. package/src/cron/.gitkeep +0 -0
  70. package/src/cron/idle-session-cron.ts +144 -0
  71. package/src/cron/idle-session-scanner.ts +37 -0
  72. package/src/cron/self-review-cron.ts +103 -0
  73. package/src/cron/semantic-consolidation.ts +228 -0
  74. package/src/memory/.gitkeep +0 -0
  75. package/src/memory/embedder.ts +15 -0
  76. package/src/memory/episodic-store.ts +183 -0
  77. package/src/memory/memory-provider.ts +98 -0
  78. package/src/memory/semantic-facts-store.ts +89 -0
  79. package/src/memory/tool-corrections-store.ts +72 -0
  80. package/src/memory/verbatim-archive-store.ts +202 -0
  81. package/src/model/client.ts +33 -0
  82. package/src/model/context-size.ts +42 -0
  83. package/src/plugins/manifest.ts +47 -0
  84. package/src/plugins/plugin-loader.ts +205 -0
  85. package/src/router/channel-loader.ts +56 -0
  86. package/src/router/provider.ts +7 -0
  87. package/src/router/terminal-provider.ts +155 -0
  88. package/src/router/terminal.ts +151 -0
  89. package/src/router/tool-log.ts +116 -0
  90. package/src/router/turn-runner.ts +205 -0
  91. package/src/session/agent-turn.ts +391 -0
  92. package/src/session/context-primer.ts +134 -0
  93. package/src/session/episodic-summarizer.ts +38 -0
  94. package/src/session/history.ts +168 -0
  95. package/src/session/pending-confirmation.ts +8 -0
  96. package/src/session/read-skill-tool.ts +38 -0
  97. package/src/session/semantic-fact-extractor.ts +69 -0
  98. package/src/session/step-info.ts +27 -0
  99. package/src/session/summarizer.ts +36 -0
  100. package/src/session/system-prompt.ts +142 -0
  101. package/src/session/tool-correction-extractor.ts +133 -0
  102. package/src/session/tool-log-buffer.ts +73 -0
  103. package/src/session/tool-log-recall-tool.ts +38 -0
  104. package/src/session/tool-start-hook.ts +164 -0
  105. package/src/tools/display-store.ts +89 -0
  106. package/src/tools/present-tool.ts +41 -0
  107. package/src/wiki/.gitkeep +0 -0
  108. package/src/wiki/frontmatter-schema.ts +49 -0
  109. package/src/wiki/index-entry.ts +59 -0
  110. package/src/wiki/orphan-detector.ts +61 -0
  111. package/src/wiki/self-review-runner.ts +133 -0
  112. package/src/wiki/self-review-tools.ts +162 -0
  113. package/src/wiki/vault-cli.ts +143 -0
  114. package/src/wiki/vault-init.ts +43 -0
  115. package/src/wiki/wiki-note.ts +326 -0
  116. package/src/wiki/wiki-read.ts +122 -0
  117. package/src/wiki/wiki-tools.ts +112 -0
@@ -0,0 +1,65 @@
1
+ import type { StepInfo } from "../session/step-info.ts";
2
+ /**
3
+ * Stringifies `value` as JSON, truncating to `maxChars` and appending a
4
+ * marker with the real total length and a pointer to `/dump` when it
5
+ * doesn't fit — so a human watching the terminal sees something bounded
6
+ * but still knows more is available and how to get it.
7
+ */
8
+ export declare function truncateForDisplay(value: unknown, maxChars: number): string;
9
+ /**
10
+ * Builds the file `/dump` writes to when the user doesn't give an
11
+ * explicit path. Lands in the OS temp dir, not the process's cwd — in
12
+ * the Docker image that's `/app`, owned by root, where the `mercury`
13
+ * user can read/execute existing files but not create new ones (every
14
+ * default-path `/dump` failed with EACCES until this). Includes a
15
+ * timestamp (colons/dots replaced since they aren't valid in filenames
16
+ * on every filesystem) so repeated `/dump` calls land in separate files
17
+ * instead of silently overwriting a fixed default each time.
18
+ */
19
+ export declare function defaultDumpPath(now?: Date): string;
20
+ /**
21
+ * Parses a `/dump [path]` command line. Returns null for anything that
22
+ * isn't exactly this command (including regular conversation input, and
23
+ * a slash-prefixed word that merely starts with "dump") — the caller
24
+ * uses this to tell a real command from a message meant for the model.
25
+ * `path` is undefined when none was given — the caller decides the
26
+ * default (see `defaultDumpPath`), since computing "now" here would make
27
+ * this function's output depend on when it happens to run.
28
+ */
29
+ export declare function parseDumpCommand(line: string): {
30
+ path: string | undefined;
31
+ } | null;
32
+ /**
33
+ * Writes `steps` to `path` as indented JSON — human-readable, since this
34
+ * is the path a person opens by hand to inspect what a tool actually
35
+ * returned, not something machine-parsed downstream.
36
+ */
37
+ export declare function writeDump(path: string, steps: StepInfo[]): Promise<void>;
38
+ /**
39
+ * Describes what happened to the tool call identified by `toolCallId`
40
+ * within `step`: its result if it executed, the `tool-error` content
41
+ * part if it failed before executing (e.g. arguments that don't match
42
+ * the tool's schema — there's no `toolResults` entry for this case, it
43
+ * only shows up in `content`), or an explicit "(none)" if neither is
44
+ * present. Printing "(none)" for an actual failure was the bug this
45
+ * fixes — it read as "nothing happened" when something had, in fact,
46
+ * gone wrong and been silently dropped from view.
47
+ */
48
+ export declare function describeToolOutcome(step: StepInfo, toolCallId: string, maxChars: number): string;
49
+ /**
50
+ * Formats real token counts for display next to the terminal prompt:
51
+ * `usedTokens` is the real `inputTokens` the last turn's call reported
52
+ * (see `src/session/agent-turn.ts`'s `onUsage`), `maxTokens` is the
53
+ * context length Ollama actually has the model loaded with right now
54
+ * (see `src/model/context-size.ts`) — not an estimate, and not the
55
+ * model's architectural maximum, which can overstate what's really
56
+ * usable. A model degrading over a long conversation is hard to tell
57
+ * apart by eye from "the context is genuinely near full" — this gives a
58
+ * live number instead of having to guess.
59
+ *
60
+ * `maxTokens` is `null` before the model has been loaded at least once
61
+ * this process (nothing to report yet) — the denominator is omitted
62
+ * rather than showing a misleading `/~0k`. `usedTokens` is `undefined`
63
+ * before the first turn has completed, shown as `?` for the same reason.
64
+ */
65
+ export declare function formatContextUsage(usedTokens: number | undefined, maxTokens: number | null): string;
@@ -0,0 +1,86 @@
1
+ /**
2
+ * Deduplicates what `src/index.ts` used to do twice: two near-identical
3
+ * ~50-line closures, one per channel, that (1) tracked the session for
4
+ * Layer-3 capture when the channel has a real per-user identity, (2) ran
5
+ * `runTurn` with a channel-specific tool set/system prompt/output sink,
6
+ * and (3) mirrored new messages to Qdrant and extracted procedural
7
+ * corrections once the turn resolved. `createTurnRunner` is that shared
8
+ * body, parameterized entirely by `Provider`/`InboundTurn`/`TurnSink`
9
+ * (`src/router/provider.ts`) so it doesn't know or care which provider a
10
+ * given turn came from — same channel-agnostic spirit as `runTurn` itself.
11
+ */
12
+ import type { LanguageModel, Tool } from "ai";
13
+ import { runTurn } from "../session/agent-turn.ts";
14
+ import type { StepInfo } from "../session/step-info.ts";
15
+ import type { PostTurnGuard } from "@mercury-fw/plugin-types";
16
+ export type { PostTurnGuard };
17
+ import type { SessionHistory } from "../session/history.ts";
18
+ import { recordStep } from "../session/tool-log-buffer.ts";
19
+ import type { HandleTurn, TurnSink } from "./provider.ts";
20
+ export type TurnRunnerDeps = {
21
+ model: LanguageModel;
22
+ /** Both variants, precomposed by the composition root; selected per turn by `turn.multiUser`. */
23
+ systemPrompts: {
24
+ singleUser: string;
25
+ multiUser: string;
26
+ };
27
+ buildTools: (sessionKey: string, wikiUserId: string, onToolStart?: TurnSink["onToolStart"], onToolFinish?: TurnSink["onToolFinish"]) => Record<string, Tool>;
28
+ /**
29
+ * `userId` is forwarded (not interpreted here) so a provider's own
30
+ * closure can decide whether to seed a first-ever session with a
31
+ * context primer (see `src/session/context-primer.ts`) — building one
32
+ * needs a real per-user identity, which only some providers have.
33
+ */
34
+ getOrCreateHistory: (sessionKey: string, trackForCapture: boolean, userId: string | undefined) => Promise<SessionHistory> | SessionHistory;
35
+ /** Layer-3 session tracking (sessionUsers map + idle scanner touch). Only for turns that carry a userId. */
36
+ trackSession: (sessionKey: string, userId: string, at: number) => void;
37
+ /** Refreshes this turn's tool-status callbacks for out-of-band capture messages. */
38
+ registerCaptureCallback: (sessionKey: string, onToolStart: TurnSink["onToolStart"], onToolFinish: TurnSink["onToolFinish"]) => void;
39
+ /** Mid-conversation Layer-3 capture threshold check. Only for turns that carry a userId. */
40
+ maybeCapture: (sessionKey: string, history: SessionHistory) => Promise<void>;
41
+ /**
42
+ * Archives one message of the verbatim user↔model exchange (issue #4).
43
+ * Wired by the composition root to the verbatim-archive provider; absent
44
+ * on an instance with no such provider. Only called for turns that carry a
45
+ * `userId` (the archive is per-user, like Layer-3 capture), keyed on the
46
+ * space-independent `wikiUserId`, and only ever with the model's own answer
47
+ * text — never the appended `present` displays.
48
+ */
49
+ captureVerbatim?: (msg: {
50
+ sessionKey: string;
51
+ userId: string;
52
+ role: "user" | "assistant";
53
+ content: string;
54
+ }) => Promise<void>;
55
+ processToolCorrections: (steps: StepInfo[], onToolStart: TurnSink["onToolStart"], onToolFinish: TurnSink["onToolFinish"]) => Promise<void>;
56
+ logStep: (prefix: string, step: StepInfo) => void;
57
+ /** Test seam; defaults to the real `recordStep`. */
58
+ recordStepFn?: typeof recordStep;
59
+ /** Test seam; defaults to the real `runTurn`. */
60
+ runTurnFn?: typeof runTurn;
61
+ /**
62
+ * Post-turn guards contributed by loaded plugins, run in order over the
63
+ * model's finished text (see `PostTurnGuard`). Empty/absent on an instance
64
+ * with no plugin that registers one. The composition root builds these; the
65
+ * core knows nothing about what any of them does.
66
+ */
67
+ postTurnGuards?: PostTurnGuard[];
68
+ /**
69
+ * Test seam; defaults to `console.log`. Receives a guard's own `log` line,
70
+ * or the core's own note when a guard throws. Exists to measure real-world
71
+ * guard frequency before investing further.
72
+ */
73
+ logPostTurnGuardFn?: (message: string) => void;
74
+ /** Test seam; defaults to `Date.now`. */
75
+ now?: () => number;
76
+ /**
77
+ * Returns the display artifacts the model surfaced via `present` this turn
78
+ * (see `display-store.ts`), in stash order, to append after the model's
79
+ * text. Absent on an instance with no display store — nothing is appended.
80
+ * The old unconditional splicing of every tool-produced display is gone:
81
+ * an artifact is shown only when the model explicitly presented it.
82
+ */
83
+ takeSurfacedDisplays?: (sessionKey: string) => string[];
84
+ };
85
+ /** Builds the shared `HandleTurn` every provider's driver calls once it has a real message to run through the model. */
86
+ export declare function createTurnRunner(deps: TurnRunnerDeps): HandleTurn;
@@ -0,0 +1,266 @@
1
+ /**
2
+ * Orchestrates a single conversational turn: append the user's input to
3
+ * history, ask the model for a response (with access to whichever tools
4
+ * this Mercury instance has wired in), append the response to history,
5
+ * return it.
6
+ *
7
+ * This is the one place that calls `generateText` for the main agent
8
+ * loop (as opposed to `src/session/summarizer.ts`, which calls it for
9
+ * summarization) — see `buildGenerateTextParams` below for why it's
10
+ * `ai-sdk-ollama`'s enhanced version, not the plain `ai` one. The real
11
+ * call is injected as
12
+ * `generateTextFn` so tests can exercise the sequencing/wiring logic
13
+ * here without needing a real model — see the test file for what that
14
+ * does and doesn't cover.
15
+ *
16
+ * Deliberately generic about *what* Mercury can do: `system` and
17
+ * `tools` are both passed in by the caller rather than hardcoded here.
18
+ * A fixed prompt baked into this file describing a specific tool (e.g.
19
+ * "use jiraCli") would be actively wrong on an instance where that tool
20
+ * isn't wired in — the model could still attempt a call to a tool name
21
+ * not present in the request's tool schema, which the AI SDK surfaces
22
+ * as a real error (`NoSuchToolError`), not a harmless no-op. Composing
23
+ * a `system` string that accurately reflects which tools are actually
24
+ * available is `src/index.ts`'s job, since that's where the set of
25
+ * enabled CLIs/tools is decided.
26
+ *
27
+ * Used by: `src/router/turn-runner.ts`'s `createTurnRunner`, the one
28
+ * shared driver every `Provider` (terminal, Google Chat) funnels its
29
+ * turns through (see `src/router/provider.ts`) — `runTurn` itself doesn't
30
+ * know or care which provider a given conversation came from.
31
+ *
32
+ * Streaming (see `buildStreamTextParams`/`runTurn`'s `onTextChunk`) is
33
+ * opt-in: providing `onTextChunk` *or* `onReasoningChunk` switches this
34
+ * call to `streamText`, reading its `fullStream` instead of waiting in
35
+ * silence for a full response — which can take several seconds on the
36
+ * local development model. The terminal provider supplies both (prints
37
+ * the answer and the model's live reasoning to stdout); Google Chat's
38
+ * `TurnSink` supplies only `onReasoningChunk` (a live-patched status card
39
+ * for the model's reasoning) and deliberately never `onTextChunk` — Chat
40
+ * only shows a message once it's fully sent, so incremental *answer*
41
+ * delivery never actually reaches a human faster there, unlike a
42
+ * terminal's live-updating stdout; the reasoning card is a different,
43
+ * already-patchable surface (see `google-chat-provider.ts`'s `createSink`).
44
+ * Either way this function still returns the full answer text and records
45
+ * it as one assistant history entry, same as the plain `generateText` path
46
+ * below — reasoning content is never part of `fullText` or the recorded
47
+ * history entry, it only ever reaches `onReasoningChunk`/`onReasoningEnd`,
48
+ * since it's a live UI-only surface, not something the model should ever
49
+ * see reflected back at it on a later turn.
50
+ */
51
+ import { type LanguageModel, type StopCondition, type Tool } from "ai";
52
+ import type { Message, SessionHistory } from "./history.ts";
53
+ import type { StepInfo } from "./step-info.ts";
54
+ import { PENDING_CONFIRMATION_NOTE } from "@mercury-fw/channel-types";
55
+ export { PENDING_CONFIRMATION_NOTE };
56
+ /** The shape of the AI SDK call this module needs, injectable for tests. */
57
+ type GenerateTextFn = (params: {
58
+ model: LanguageModel;
59
+ messages: Message[];
60
+ tools: Record<string, Tool>;
61
+ instructions: string;
62
+ onStepEnd?: (step: StepInfo) => void;
63
+ /** Aborts the in-flight generation when the caller cancels the turn. */
64
+ abortSignal?: AbortSignal;
65
+ }) => Promise<{
66
+ text: string;
67
+ usage?: {
68
+ inputTokens: number | undefined;
69
+ };
70
+ }>;
71
+ /**
72
+ * Minimal shape this file reads off a `fullStream` part — deliberately a
73
+ * loose supertype of the real (much larger) `TextStreamPart` union `ai`
74
+ * actually emits (`node_modules/ai/dist/index.d.ts:2484`: `start`,
75
+ * `text-start`, `tool-call`, `finish`, etc.), not a closed enumeration of
76
+ * it: a closed union here would reject every part type this file doesn't
77
+ * care about, which the real stream emits plenty of. `delta`/`text` are
78
+ * both optional for the same reason — only present on the part types this
79
+ * file actually reads (`text-delta`/`reasoning-delta`), and the *name* of
80
+ * that field is itself unstable: `ai`'s own `.d.ts` declares this part
81
+ * shape twice with different field names for the same `type` value
82
+ * (`node_modules/ai/dist/index.d.ts:2103-2107` uses `delta`, `2555-2558`
83
+ * uses `text`) — confirmed live against the real `ai-sdk-ollama` stream
84
+ * that the installed version actually emits `text`, not `delta`. Reading
85
+ * both defensively (see the loop below) survives either shape rather than
86
+ * silently reading `undefined` and going empty if a future upgrade flips
87
+ * it back.
88
+ */
89
+ type StreamPart = {
90
+ type: string;
91
+ delta?: string;
92
+ text?: string;
93
+ id?: string;
94
+ };
95
+ /** The shape of the AI SDK streaming call this module needs, injectable for tests. */
96
+ type StreamTextFn = (params: {
97
+ model: LanguageModel;
98
+ messages: Message[];
99
+ tools: Record<string, Tool>;
100
+ instructions: string;
101
+ onStepEnd?: (step: StepInfo) => void;
102
+ /** Aborts the in-flight generation when the caller cancels the turn. */
103
+ abortSignal?: AbortSignal;
104
+ }) => Promise<{
105
+ stream: AsyncIterable<StreamPart>;
106
+ usage?: PromiseLike<{
107
+ inputTokens: number | undefined;
108
+ }>;
109
+ }>;
110
+ /**
111
+ * Builds the params object passed to the real `generateText`. Extracted
112
+ * as its own pure function so it's unit-testable without a real model —
113
+ * `defaultGenerateTextFn` below is otherwise a thin, untestable wrapper
114
+ * around a direct SDK call.
115
+ *
116
+ * `stopWhen: stepCountIs(100)` is the one non-obvious part: `generateText`
117
+ * defaults to stopping after a single step. If that step is a tool call
118
+ * with no accompanying text (the normal case — the model calls a tool,
119
+ * then needs the tool's result before it can answer), the call returns
120
+ * with `text: ""` and no error, having never given the model a chance to
121
+ * read the tool result and respond — some cap above 1 is needed for
122
+ * that. Raised from an original 5 (observed live: a conversation
123
+ * combining wiki lookups with a jira search — which always needs
124
+ * `--select`, so almost always costs 2 attempts — routinely burned all 5
125
+ * steps on tool calls alone, leaving the model zero steps to ever write
126
+ * an answer) to 20 (matching this SDK's own default for its
127
+ * higher-level agent construct, `ToolLoopAgentSettings`), then to 100 for
128
+ * extra headroom during this research phase — the model should be free
129
+ * to make as many tool calls as it actually needs, not be cut off by an
130
+ * arbitrary ceiling. `runTurn`'s empty-text fallback (below) is the
131
+ * remaining safety net regardless of the cap's value: if a turn still
132
+ * exhausts it with nothing to show, that's communicated explicitly
133
+ * rather than left silent — the real backstop against a runaway loop is
134
+ * that fallback plus the step count itself still being finite, not the
135
+ * specific number.
136
+ *
137
+ * `generateText` itself is imported from `ai-sdk-ollama`, not the plain
138
+ * `ai` package — confirmed by a real run: with the standard SDK's
139
+ * version, Ollama executed the tool call but `text` still came back
140
+ * empty even with multi-step enabled. `ai-sdk-ollama`'s enhanced
141
+ * `generateText` is a drop-in replacement (same params/return shape)
142
+ * that specifically synthesizes a real response when this happens —
143
+ * documented as a known Ollama-provider quirk, not something to patch
144
+ * around here by hand.
145
+ */
146
+ export declare function buildGenerateTextParams(params: {
147
+ model: LanguageModel;
148
+ messages: Message[];
149
+ tools: Record<string, Tool>;
150
+ instructions: string;
151
+ onStepEnd?: (step: StepInfo) => void;
152
+ abortSignal?: AbortSignal;
153
+ }): {
154
+ stopWhen: StopCondition<any>[];
155
+ model: LanguageModel;
156
+ messages: Message[];
157
+ tools: Record<string, Tool>;
158
+ instructions: string;
159
+ onStepEnd?: (step: StepInfo) => void;
160
+ abortSignal?: AbortSignal;
161
+ };
162
+ /**
163
+ * Builds the params object passed to the real `streamText`. Same
164
+ * `stopWhen` reasoning as `buildGenerateTextParams` — a tool-call-only
165
+ * first step must not be the stream's last step either. `ai-sdk-ollama`'s
166
+ * `streamText` carries its own equivalent of the empty-text-after-
167
+ * tool-call fix (`enableStreamingSynthesis`, on by default) — verified by
168
+ * reading its source before relying on it, not assumed from the
169
+ * non-streaming behavior.
170
+ */
171
+ export declare function buildStreamTextParams(params: {
172
+ model: LanguageModel;
173
+ messages: Message[];
174
+ tools: Record<string, Tool>;
175
+ instructions: string;
176
+ onStepEnd?: (step: StepInfo) => void;
177
+ abortSignal?: AbortSignal;
178
+ }): {
179
+ stopWhen: StopCondition<any>[];
180
+ model: LanguageModel;
181
+ messages: Message[];
182
+ tools: Record<string, Tool>;
183
+ instructions: string;
184
+ onStepEnd?: (step: StepInfo) => void;
185
+ abortSignal?: AbortSignal;
186
+ };
187
+ /**
188
+ * Runs one turn: records `userInput`, generates a response with the
189
+ * given model/tools/system prompt, records the response, and returns it.
190
+ *
191
+ * @param history - The conversation's `SessionHistory`; one per
192
+ * channel/space, never shared (see `src/index.ts`).
193
+ * @param userInput - The user's message for this turn.
194
+ * @param deps.model - The language model to use.
195
+ * @param deps.tools - The tools available to the model on this call —
196
+ * which tools end up here depends on which CLIs this Mercury instance
197
+ * has enabled (see `src/index.ts`), not on anything in this file.
198
+ * @param deps.system - The system prompt for this call. Must accurately
199
+ * describe only the tools actually present in `deps.tools` — this
200
+ * function doesn't validate that, the caller is responsible for
201
+ * keeping the two in sync.
202
+ * @param deps.onStepFinish - Optional, called once per generation step
203
+ * (including intermediate ones with only a tool call, no text).
204
+ * `src/router/turn-runner.ts`'s shared `createTurnRunner` wires this for
205
+ * every provider (fans out to `logStep`/`recordStep`/the provider's own
206
+ * `TurnSink.onStep`) — what each provider actually *does* with a step
207
+ * still differs (the terminal prints it, Google Chat's `TurnSink`
208
+ * doesn't define `onStep` at all, showing raw tool calls to a chat
209
+ * audience isn't the same call as showing them to whoever's debugging at
210
+ * a terminal), but the wiring itself is no longer channel-specific.
211
+ * @param deps.onTextChunk - Optional. When provided (alongside or instead
212
+ * of `onReasoningChunk`), this turn uses `streamText` instead of
213
+ * `generateText`, calling this once per answer-text chunk as it arrives
214
+ * — see the file header for why Google Chat never sets this one. The
215
+ * returned string and the recorded history entry are the same either
216
+ * way: the full answer text, joined from every `text-delta` chunk only.
217
+ * @param deps.onReasoningChunk - Optional. Also switches this turn onto
218
+ * `streamText` (see `onTextChunk`). Called once per reasoning-token
219
+ * delta as it streams, tagged with the SDK's own id for that reasoning
220
+ * block — only ever fires for a model that actually supports Ollama's
221
+ * native extended thinking (see `src/index.ts`'s `OLLAMA_THINK`);
222
+ * otherwise no reasoning parts ever arrive and this is simply never
223
+ * called. This content is UI-only: it never touches the returned text
224
+ * or `SessionHistory`. A single turn can reason more than once (e.g.
225
+ * once before a tool call, again after seeing its result) — each burst
226
+ * carries its own id, letting a caller (e.g. a status card per id)
227
+ * treat them as independent rather than one continuous stream.
228
+ * @param deps.onReasoningEnd - Optional. Fires once per reasoning block
229
+ * that actually started (i.e. once per distinct id `onReasoningChunk`
230
+ * reported) — including if the stream aborts while a block is still
231
+ * open, so a caller building a live display (a status card, a printed
232
+ * block) can't be left stuck open forever. `failed` is `true` only for
233
+ * that abrupt-abort case, `false` on a normal reasoning-end.
234
+ * @param deps.generateTextFn - Test seam for the non-streaming path;
235
+ * defaults to the real AI SDK call. Injecting a fake here only tests
236
+ * this function's own sequencing — it does not exercise the real model
237
+ * or the real AI SDK integration, which can only be verified by an
238
+ * actual end-to-end run.
239
+ * @param deps.streamTextFn - Test seam for the streaming path (used when
240
+ * `onTextChunk`/`onReasoningChunk` is provided), same caveat as
241
+ * `generateTextFn`.
242
+ * @param deps.onUsage - Optional, called once per turn with the real
243
+ * `inputTokens` count the model actually consumed (summed across every
244
+ * step, including tool calls) — not an estimate. The terminal channel
245
+ * uses this for a real context-usage indicator (see
246
+ * `src/router/tool-log.ts`'s `formatContextUsage`); Mercury's own char-
247
+ * count heuristic in `src/session/history.ts` is for a different
248
+ * purpose (deciding when to summarize) and intentionally untouched by
249
+ * this.
250
+ */
251
+ export declare function runTurn(history: SessionHistory, userInput: string, deps: {
252
+ model: LanguageModel;
253
+ tools: Record<string, Tool>;
254
+ system: string;
255
+ onStepFinish?: (step: StepInfo) => void;
256
+ onTextChunk?: (chunk: string) => void;
257
+ onReasoningChunk?: (chunk: string, id: string) => void;
258
+ onReasoningEnd?: (id: string, failed: boolean) => void;
259
+ onUsage?: (inputTokens: number | undefined) => void;
260
+ generateTextFn?: GenerateTextFn;
261
+ streamTextFn?: StreamTextFn;
262
+ /** When set, aborting it stops the in-flight generation — the turn ends
263
+ * promptly instead of running to completion (see the HTTP surface's
264
+ * client-disconnect cancellation). Optional; existing callers pass none. */
265
+ abortSignal?: AbortSignal;
266
+ }): Promise<string>;
@@ -0,0 +1,16 @@
1
+ import type { EpisodicSummary } from "../memory/episodic-store.ts";
2
+ import type { listWikiFilesInRoots, readWikiFileInRoots, readIndexFile } from "../wiki/wiki-read.ts";
3
+ export type ContextPrimerDeps = {
4
+ vaultPath: string;
5
+ /** Already scoped to the last closed session's own sessionKey — see `getLastSessionEpisodicSummaries`. */
6
+ getLastSessionEntries: (userId: string) => Promise<EpisodicSummary[]>;
7
+ listWikiFilesInRootsFn: typeof listWikiFilesInRoots;
8
+ readWikiFileInRootsFn: typeof readWikiFileInRoots;
9
+ readIndexFileFn: typeof readIndexFile;
10
+ };
11
+ /**
12
+ * Text of the primer for `userId`, built from injected deps only — never
13
+ * touches Qdrant or the filesystem directly, so tests supply fakes and
14
+ * `index.ts` supplies the real Qdrant-backed episodic query and wiki reads.
15
+ */
16
+ export declare function buildContextPrimer(userId: string, deps: ContextPrimerDeps): Promise<string>;
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Thin glue turning a closed session's messages into the episodic
3
+ * summary written to Qdrant (see `src/memory/episodic-store.ts`) — a
4
+ * factual account of what happened, not an interpretation. Distinct from
5
+ * `summarizer.ts` (Layer 1): that one condenses history to keep the
6
+ * *next* turn's prompt small, preserving whatever helps continue the
7
+ * same conversation; this one produces a standalone record of a
8
+ * *finished* conversation, and must not infer patterns or preferences
9
+ * (that inference is a separate, deterministic consolidation step,
10
+ * not this LLM call's job).
11
+ *
12
+ * Deliberately never asks the model for a date: it has no reliable notion
13
+ * of "today" and would invent one (observed live: dates from the wrong
14
+ * year, hedged with "replace with current date if applicable"). The
15
+ * entry's own `timestamp` field (computed by the caller,
16
+ * `idle-session-cron.ts`) is the one and only source of truth for "when" —
17
+ * never duplicated into this text, mechanically or otherwise.
18
+ *
19
+ * Same "not worth mocking deeply" reasoning as `summarizer.ts` — no
20
+ * dedicated test file, it's one line of glue around `generateText`.
21
+ */
22
+ import { type LanguageModel } from "ai";
23
+ import type { Message } from "./history.ts";
24
+ /** Returns a function that summarizes a closed session's messages into a factual account. */
25
+ export declare function createEpisodicSummarizer(model: LanguageModel): (messages: Message[]) => Promise<string>;
@@ -0,0 +1,95 @@
1
+ /**
2
+ * Layer 1 conversation memory: a sliding window of raw messages that
3
+ * summarizes itself once it grows too large, instead of growing
4
+ * unbounded across a long conversation.
5
+ *
6
+ * Why it exists: without a bound, a multi-turn conversation eventually
7
+ * overflows the model's context window. This is the only memory layer
8
+ * Mercury has in M1 — Layer 2 (wiki) and Layer 3 (episodic/Qdrant) are
9
+ * later milestones, pure enrichment that the system must work without.
10
+ *
11
+ * Used by: `src/session/agent-turn.ts` (`runTurn`), which appends each
12
+ * turn's user/assistant messages here and reads `getMessages()` to build
13
+ * the prompt for the next generation call. `src/session/summarizer.ts`
14
+ * supplies the `summarize` function injected into `createSessionHistory`.
15
+ */
16
+ /** A single turn's worth of conversation content. */
17
+ export type Message = {
18
+ role: "user" | "assistant";
19
+ content: string;
20
+ };
21
+ /**
22
+ * Character-count threshold (not a real tokenizer count — a `chars/4`
23
+ * estimate is close enough given the wide margin in the model's context
24
+ * budget) above which the raw message history gets summarized and
25
+ * replaced. Exported so tests can construct fixtures that land exactly
26
+ * at, or just past, the boundary.
27
+ */
28
+ export declare const MAX_HISTORY_CHARS = 60000;
29
+ /**
30
+ * Mutable conversation history for a single ongoing conversation
31
+ * (one per channel/space — see `src/index.ts`, never shared across
32
+ * conversations).
33
+ */
34
+ export type SessionHistory = {
35
+ /** Appends a user turn, summarizing first if this push crosses the threshold. */
36
+ addUserMessage(content: string): Promise<void>;
37
+ /** Appends an assistant turn, summarizing first if this push crosses the threshold. */
38
+ addAssistantMessage(content: string): Promise<void>;
39
+ /**
40
+ * Overwrites the most recently added message with `content`, in place,
41
+ * if it exists and is an assistant message — used when a turn's
42
+ * assistant text is corrected after already being recorded (see
43
+ * turn-runner.ts's issue-list correction). No-ops if there is no last
44
+ * message, or if it isn't an assistant message (defensive; shouldn't
45
+ * happen given how this is called). Deliberately synchronous and skips
46
+ * the summarization threshold check entirely: this replaces content
47
+ * already counted by the original `addAssistantMessage` call, it isn't
48
+ * new content being added.
49
+ */
50
+ replaceLastAssistantMessage(content: string): void;
51
+ /**
52
+ * The messages to feed into the next model call: the current summary
53
+ * (if one exists, as a synthetic leading message) followed by the raw
54
+ * messages accumulated since the last summarization.
55
+ */
56
+ getMessages(): Message[];
57
+ /**
58
+ * Total character length of what `getMessages()` would currently
59
+ * return — a live read on how close this conversation is to
60
+ * `MAX_HISTORY_CHARS` (and so to triggering summarization). Exposed so
61
+ * a channel can show this to a human, e.g. to tell apart "the model
62
+ * lost track of something" from "the context is actually near full".
63
+ */
64
+ getCharCount(): number;
65
+ };
66
+ /**
67
+ * Creates an empty `SessionHistory`.
68
+ *
69
+ * @param summarize - Called with the messages that precede the current
70
+ * user turn (any prior summary re-injected as a leading message) whenever
71
+ * a single append pushes the total content length over
72
+ * `MAX_HISTORY_CHARS`. Its return value becomes the new summary; the
73
+ * trailing run from the last user message onward is retained as the raw
74
+ * window rather than cleared, so the current turn is never summarized away
75
+ * (a model call always follows `addUserMessage`, and the primer/summary
76
+ * leading messages are both `role:"assistant"` — folding the user turn
77
+ * into them would hand the model a user-less array). The threshold check
78
+ * runs after every individual append (not once per turn), so the crossing
79
+ * point is caught precisely regardless of whether it's the user or
80
+ * assistant message that tips it over. A lone crossing user message with
81
+ * nothing before it is left live and `summarize` is not called.
82
+ * @param onBeforeCompress - Optional, called synchronously with the exact
83
+ * same batch `summarize` is about to receive, right before it's
84
+ * compressed out of the live context — a second, independent signal a
85
+ * caller can mirror to somewhere durable (see `idle-session-cron.ts`'s
86
+ * shared capture function) before that content stops being directly
87
+ * visible to the model. Fire-and-forget on purpose: this function must
88
+ * never block or fail Layer 1's own compression on an external write.
89
+ * @param primer - Optional, set once at creation from the user's last closed
90
+ * session (see `context-primer.ts`). Held as its own state, entirely
91
+ * independent from `summary`: it's never included in the batch passed to
92
+ * `summarize`, so a real compression event can't paraphrase or drop it.
93
+ * Leads `getMessages()` for the whole life of this history.
94
+ */
95
+ export declare function createSessionHistory(summarize: (messages: Message[]) => Promise<string>, onBeforeCompress?: (messages: Message[]) => void, primer?: string): SessionHistory;
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Confirm-required detection, re-exported from `@mercury-fw/channel-types`. The
3
+ * logic moved to the shared package so channel plugins can import it without
4
+ * depending on the app; kept re-exported here for the core callers
5
+ * (`agent-turn.ts` and the tests) that import it from this path.
6
+ */
7
+ export { detectPendingConfirmation } from "@mercury-fw/channel-types";
8
+ export type { PendingConfirmation } from "@mercury-fw/channel-types";
@@ -0,0 +1,4 @@
1
+ import type { Skill, ExecutableTool } from "@mercury-fw/plugin-types";
2
+ export declare function createReadSkillTool(skills: Skill[]): {
3
+ read_skill: ExecutableTool;
4
+ };
@@ -0,0 +1,45 @@
1
+ /**
2
+ * Turns a closed session's messages into structured `{topic, value}`
3
+ * facts for the semantic consolidation engine (see
4
+ * `src/memory/semantic-facts-store.ts`) — distinct from
5
+ * `episodic-summarizer.ts`, which produces a prose account of the whole
6
+ * session. A single extracted fact here is a candidate, not yet a
7
+ * standing belief about the user: consolidation (a separate,
8
+ * deterministic step) decides whether repeated facts on the same topic
9
+ * are frequent enough to be promoted to a wiki note.
10
+ */
11
+ import { type LanguageModel } from "ai";
12
+ import { z } from "zod";
13
+ import type { Message } from "./history.ts";
14
+ /**
15
+ * Closed vocabulary — the model can only ever return one of these exact
16
+ * values, never invent a new key for the same concept. Deliberately
17
+ * excludes identity/name: a registered Chat app's own `MESSAGE` event
18
+ * already carries the sender's `displayName` directly, so a semantic
19
+ * fact about "who the user is" would only duplicate or contradict that
20
+ * more authoritative source, never add anything — observed live as the
21
+ * `name`/`user-name` duplicate before this fix.
22
+ */
23
+ export declare const SEMANTIC_FACT_TOPICS: readonly ["team", "role", "preferred-language", "tools-used"];
24
+ export declare const SemanticFactSchema: z.ZodObject<{
25
+ topic: z.ZodEnum<{
26
+ team: "team";
27
+ role: "role";
28
+ "preferred-language": "preferred-language";
29
+ "tools-used": "tools-used";
30
+ }>;
31
+ value: z.ZodString;
32
+ }, z.core.$strip>;
33
+ export type SemanticFact = z.infer<typeof SemanticFactSchema>;
34
+ type GenerateObjectFn = (params: {
35
+ model: LanguageModel;
36
+ output: "array";
37
+ schema: typeof SemanticFactSchema;
38
+ instructions: string;
39
+ prompt: string;
40
+ }) => Promise<{
41
+ object: SemanticFact[];
42
+ }>;
43
+ /** Returns a function that extracts standing {topic, value} facts from a closed session's messages. */
44
+ export declare function createSemanticFactExtractor(model: LanguageModel, generateObjectFn?: GenerateObjectFn): (messages: Message[]) => Promise<SemanticFact[]>;
45
+ export {};
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Minimal shape of a finished generation step that callers might care
3
+ * about — just enough to show what tool Mercury called, with what
4
+ * input, and what it got back. `toolCallId` is what links an entry in
5
+ * `toolCalls` to its entry in `toolResults` — a call with no matching
6
+ * result is a real case callers need to handle explicitly rather than
7
+ * assume a 1:1 pairing: it means the call failed before ever executing
8
+ * (e.g. malformed arguments that don't match the tool's schema), which
9
+ * shows up as a `tool-error` entry in `content`, not in `toolResults` —
10
+ * confirmed against the real AI SDK's `StepResult` type, which has no
11
+ * separate `toolErrors` array; `content` is the one place every part
12
+ * (text/tool-call/tool-result/tool-error) actually lives. The real AI
13
+ * SDK step object has many more fields; this is a subset, which is fine
14
+ * since function parameter types only need to be structurally
15
+ * compatible, not identical.
16
+ *
17
+ * Lives in its own file, not `agent-turn.ts`, specifically so
18
+ * `pending-confirmation.ts` (which needs this type) and `agent-turn.ts`
19
+ * (which needs `pending-confirmation.ts`'s `detectPendingConfirmation` to
20
+ * decide whether to stop the tool-calling loop early) don't form an
21
+ * import cycle.
22
+ */
23
+ import type { StepInfo } from "@mercury-fw/plugin-types";
24
+ export type { StepInfo };