@mercury-fw/core 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/README.md +38 -0
  3. package/dist/index.d.ts +23 -0
  4. package/dist/src/admin/cli-routes.d.ts +22 -0
  5. package/dist/src/admin/env-file.d.ts +1 -0
  6. package/dist/src/admin/model-routes.d.ts +26 -0
  7. package/dist/src/admin/qdrant-scroll.d.ts +34 -0
  8. package/dist/src/admin/server.d.ts +40 -0
  9. package/dist/src/admin/wiki-routes.d.ts +31 -0
  10. package/dist/src/compose.d.ts +42 -0
  11. package/dist/src/config/define-config.d.ts +31 -0
  12. package/dist/src/cron/idle-session-cron.d.ts +80 -0
  13. package/dist/src/cron/idle-session-scanner.d.ts +16 -0
  14. package/dist/src/cron/self-review-cron.d.ts +55 -0
  15. package/dist/src/cron/semantic-consolidation.d.ts +71 -0
  16. package/dist/src/memory/embedder.d.ts +9 -0
  17. package/dist/src/memory/episodic-store.d.ts +121 -0
  18. package/dist/src/memory/memory-provider.d.ts +51 -0
  19. package/dist/src/memory/semantic-facts-store.d.ts +37 -0
  20. package/dist/src/memory/tool-corrections-store.d.ts +26 -0
  21. package/dist/src/memory/verbatim-archive-store.d.ts +86 -0
  22. package/dist/src/model/client.d.ts +24 -0
  23. package/dist/src/model/context-size.d.ts +30 -0
  24. package/dist/src/plugins/manifest.d.ts +29 -0
  25. package/dist/src/plugins/plugin-loader.d.ts +85 -0
  26. package/dist/src/router/channel-loader.d.ts +30 -0
  27. package/dist/src/router/provider.d.ts +7 -0
  28. package/dist/src/router/terminal-provider.d.ts +37 -0
  29. package/dist/src/router/terminal.d.ts +41 -0
  30. package/dist/src/router/tool-log.d.ts +65 -0
  31. package/dist/src/router/turn-runner.d.ts +86 -0
  32. package/dist/src/session/agent-turn.d.ts +266 -0
  33. package/dist/src/session/context-primer.d.ts +16 -0
  34. package/dist/src/session/episodic-summarizer.d.ts +25 -0
  35. package/dist/src/session/history.d.ts +95 -0
  36. package/dist/src/session/pending-confirmation.d.ts +8 -0
  37. package/dist/src/session/read-skill-tool.d.ts +4 -0
  38. package/dist/src/session/semantic-fact-extractor.d.ts +45 -0
  39. package/dist/src/session/step-info.d.ts +24 -0
  40. package/dist/src/session/summarizer.d.ts +23 -0
  41. package/dist/src/session/system-prompt.d.ts +38 -0
  42. package/dist/src/session/tool-correction-extractor.d.ts +43 -0
  43. package/dist/src/session/tool-log-buffer.d.ts +24 -0
  44. package/dist/src/session/tool-log-recall-tool.d.ts +18 -0
  45. package/dist/src/session/tool-start-hook.d.ts +57 -0
  46. package/dist/src/tools/display-store.d.ts +36 -0
  47. package/dist/src/tools/present-tool.d.ts +23 -0
  48. package/dist/src/wiki/frontmatter-schema.d.ts +53 -0
  49. package/dist/src/wiki/index-entry.d.ts +15 -0
  50. package/dist/src/wiki/orphan-detector.d.ts +1 -0
  51. package/dist/src/wiki/self-review-runner.d.ts +48 -0
  52. package/dist/src/wiki/self-review-tools.d.ts +22 -0
  53. package/dist/src/wiki/vault-cli.d.ts +2 -0
  54. package/dist/src/wiki/vault-init.d.ts +7 -0
  55. package/dist/src/wiki/wiki-note.d.ts +62 -0
  56. package/dist/src/wiki/wiki-read.d.ts +27 -0
  57. package/dist/src/wiki/wiki-tools.d.ts +7 -0
  58. package/index.ts +23 -0
  59. package/package.json +49 -0
  60. package/src/admin/cli-routes.ts +48 -0
  61. package/src/admin/env-file.ts +29 -0
  62. package/src/admin/model-routes.ts +71 -0
  63. package/src/admin/public/index.html +416 -0
  64. package/src/admin/qdrant-scroll.ts +45 -0
  65. package/src/admin/server.ts +188 -0
  66. package/src/admin/wiki-routes.ts +93 -0
  67. package/src/compose.ts +599 -0
  68. package/src/config/define-config.ts +35 -0
  69. package/src/cron/.gitkeep +0 -0
  70. package/src/cron/idle-session-cron.ts +144 -0
  71. package/src/cron/idle-session-scanner.ts +37 -0
  72. package/src/cron/self-review-cron.ts +103 -0
  73. package/src/cron/semantic-consolidation.ts +228 -0
  74. package/src/memory/.gitkeep +0 -0
  75. package/src/memory/embedder.ts +15 -0
  76. package/src/memory/episodic-store.ts +183 -0
  77. package/src/memory/memory-provider.ts +98 -0
  78. package/src/memory/semantic-facts-store.ts +89 -0
  79. package/src/memory/tool-corrections-store.ts +72 -0
  80. package/src/memory/verbatim-archive-store.ts +202 -0
  81. package/src/model/client.ts +33 -0
  82. package/src/model/context-size.ts +42 -0
  83. package/src/plugins/manifest.ts +47 -0
  84. package/src/plugins/plugin-loader.ts +205 -0
  85. package/src/router/channel-loader.ts +56 -0
  86. package/src/router/provider.ts +7 -0
  87. package/src/router/terminal-provider.ts +155 -0
  88. package/src/router/terminal.ts +151 -0
  89. package/src/router/tool-log.ts +116 -0
  90. package/src/router/turn-runner.ts +205 -0
  91. package/src/session/agent-turn.ts +391 -0
  92. package/src/session/context-primer.ts +134 -0
  93. package/src/session/episodic-summarizer.ts +38 -0
  94. package/src/session/history.ts +168 -0
  95. package/src/session/pending-confirmation.ts +8 -0
  96. package/src/session/read-skill-tool.ts +38 -0
  97. package/src/session/semantic-fact-extractor.ts +69 -0
  98. package/src/session/step-info.ts +27 -0
  99. package/src/session/summarizer.ts +36 -0
  100. package/src/session/system-prompt.ts +142 -0
  101. package/src/session/tool-correction-extractor.ts +133 -0
  102. package/src/session/tool-log-buffer.ts +73 -0
  103. package/src/session/tool-log-recall-tool.ts +38 -0
  104. package/src/session/tool-start-hook.ts +164 -0
  105. package/src/tools/display-store.ts +89 -0
  106. package/src/tools/present-tool.ts +41 -0
  107. package/src/wiki/.gitkeep +0 -0
  108. package/src/wiki/frontmatter-schema.ts +49 -0
  109. package/src/wiki/index-entry.ts +59 -0
  110. package/src/wiki/orphan-detector.ts +61 -0
  111. package/src/wiki/self-review-runner.ts +133 -0
  112. package/src/wiki/self-review-tools.ts +162 -0
  113. package/src/wiki/vault-cli.ts +143 -0
  114. package/src/wiki/vault-init.ts +43 -0
  115. package/src/wiki/wiki-note.ts +326 -0
  116. package/src/wiki/wiki-read.ts +122 -0
  117. package/src/wiki/wiki-tools.ts +112 -0
@@ -0,0 +1,168 @@
1
+ /**
2
+ * Layer 1 conversation memory: a sliding window of raw messages that
3
+ * summarizes itself once it grows too large, instead of growing
4
+ * unbounded across a long conversation.
5
+ *
6
+ * Why it exists: without a bound, a multi-turn conversation eventually
7
+ * overflows the model's context window. This is the only memory layer
8
+ * Mercury has in M1 — Layer 2 (wiki) and Layer 3 (episodic/Qdrant) are
9
+ * later milestones, pure enrichment that the system must work without.
10
+ *
11
+ * Used by: `src/session/agent-turn.ts` (`runTurn`), which appends each
12
+ * turn's user/assistant messages here and reads `getMessages()` to build
13
+ * the prompt for the next generation call. `src/session/summarizer.ts`
14
+ * supplies the `summarize` function injected into `createSessionHistory`.
15
+ */
16
+
17
+ /** A single turn's worth of conversation content. */
18
+ export type Message = { role: "user" | "assistant"; content: string };
19
+
20
+ /**
21
+ * Character-count threshold (not a real tokenizer count — a `chars/4`
22
+ * estimate is close enough given the wide margin in the model's context
23
+ * budget) above which the raw message history gets summarized and
24
+ * replaced. Exported so tests can construct fixtures that land exactly
25
+ * at, or just past, the boundary.
26
+ */
27
+ export const MAX_HISTORY_CHARS = 60_000;
28
+
29
+ /**
30
+ * Mutable conversation history for a single ongoing conversation
31
+ * (one per channel/space — see `src/index.ts`, never shared across
32
+ * conversations).
33
+ */
34
+ export type SessionHistory = {
35
+ /** Appends a user turn, summarizing first if this push crosses the threshold. */
36
+ addUserMessage(content: string): Promise<void>;
37
+ /** Appends an assistant turn, summarizing first if this push crosses the threshold. */
38
+ addAssistantMessage(content: string): Promise<void>;
39
+ /**
40
+ * Overwrites the most recently added message with `content`, in place,
41
+ * if it exists and is an assistant message — used when a turn's
42
+ * assistant text is corrected after already being recorded (see
43
+ * turn-runner.ts's issue-list correction). No-ops if there is no last
44
+ * message, or if it isn't an assistant message (defensive; shouldn't
45
+ * happen given how this is called). Deliberately synchronous and skips
46
+ * the summarization threshold check entirely: this replaces content
47
+ * already counted by the original `addAssistantMessage` call, it isn't
48
+ * new content being added.
49
+ */
50
+ replaceLastAssistantMessage(content: string): void;
51
+ /**
52
+ * The messages to feed into the next model call: the current summary
53
+ * (if one exists, as a synthetic leading message) followed by the raw
54
+ * messages accumulated since the last summarization.
55
+ */
56
+ getMessages(): Message[];
57
+ /**
58
+ * Total character length of what `getMessages()` would currently
59
+ * return — a live read on how close this conversation is to
60
+ * `MAX_HISTORY_CHARS` (and so to triggering summarization). Exposed so
61
+ * a channel can show this to a human, e.g. to tell apart "the model
62
+ * lost track of something" from "the context is actually near full".
63
+ */
64
+ getCharCount(): number;
65
+ };
66
+
67
+ /** Wraps a summary string as the synthetic leading message `getMessages()` prepends. */
68
+ function summaryMessage(summary: string): Message {
69
+ return { role: "assistant", content: `Earlier conversation summary: ${summary}` };
70
+ }
71
+
72
+ /**
73
+ * Wraps a primer string as the synthetic leading message `getMessages()`
74
+ * prepends ahead of any `summary` message. Distinct wording from
75
+ * `summaryMessage` on purpose — the primer describes the user's last
76
+ * *closed* session, not a summary of *this* one, and must not be confused
77
+ * with it.
78
+ */
79
+ function primerMessage(primer: string): Message {
80
+ return { role: "assistant", content: `Context from your last session: ${primer}` };
81
+ }
82
+
83
+ /**
84
+ * Creates an empty `SessionHistory`.
85
+ *
86
+ * @param summarize - Called with the messages that precede the current
87
+ * user turn (any prior summary re-injected as a leading message) whenever
88
+ * a single append pushes the total content length over
89
+ * `MAX_HISTORY_CHARS`. Its return value becomes the new summary; the
90
+ * trailing run from the last user message onward is retained as the raw
91
+ * window rather than cleared, so the current turn is never summarized away
92
+ * (a model call always follows `addUserMessage`, and the primer/summary
93
+ * leading messages are both `role:"assistant"` — folding the user turn
94
+ * into them would hand the model a user-less array). The threshold check
95
+ * runs after every individual append (not once per turn), so the crossing
96
+ * point is caught precisely regardless of whether it's the user or
97
+ * assistant message that tips it over. A lone crossing user message with
98
+ * nothing before it is left live and `summarize` is not called.
99
+ * @param onBeforeCompress - Optional, called synchronously with the exact
100
+ * same batch `summarize` is about to receive, right before it's
101
+ * compressed out of the live context — a second, independent signal a
102
+ * caller can mirror to somewhere durable (see `idle-session-cron.ts`'s
103
+ * shared capture function) before that content stops being directly
104
+ * visible to the model. Fire-and-forget on purpose: this function must
105
+ * never block or fail Layer 1's own compression on an external write.
106
+ * @param primer - Optional, set once at creation from the user's last closed
107
+ * session (see `context-primer.ts`). Held as its own state, entirely
108
+ * independent from `summary`: it's never included in the batch passed to
109
+ * `summarize`, so a real compression event can't paraphrase or drop it.
110
+ * Leads `getMessages()` for the whole life of this history.
111
+ */
112
+ export function createSessionHistory(
113
+ summarize: (messages: Message[]) => Promise<string>,
114
+ onBeforeCompress?: (messages: Message[]) => void,
115
+ primer?: string,
116
+ ): SessionHistory {
117
+ let rawMessages: Message[] = [];
118
+ let summary: string | null = null;
119
+
120
+ async function add(message: Message): Promise<void> {
121
+ rawMessages.push(message);
122
+
123
+ const total = rawMessages.reduce((sum, m) => sum + m.content.length, 0);
124
+ if (total > MAX_HISTORY_CHARS) {
125
+ // Retain the trailing run from the last user message onward instead of
126
+ // clearing everything: getMessages() for a model call always follows an
127
+ // addUserMessage, and the primer/summary leading messages are both
128
+ // role:"assistant", so summarizing the current user turn away would hand
129
+ // Ollama a user-less array ("no user query found in messages") — bug #19.
130
+ // Compress only what precedes that turn; if nothing precedes it (a lone
131
+ // oversized user message), leave it live — its question can't be summarized.
132
+ const lastUserIdx = rawMessages.findLastIndex((m) => m.role === "user");
133
+ const toCompress = lastUserIdx >= 0 ? rawMessages.slice(0, lastUserIdx) : rawMessages;
134
+ const retained = lastUserIdx >= 0 ? rawMessages.slice(lastUserIdx) : [];
135
+ if (toCompress.length === 0) {
136
+ return;
137
+ }
138
+ const batch = summary ? [summaryMessage(summary), ...toCompress] : toCompress;
139
+ onBeforeCompress?.(batch);
140
+ summary = await summarize(batch);
141
+ rawMessages = retained;
142
+ }
143
+ }
144
+
145
+ function getMessages(): Message[] {
146
+ const leading: Message[] = [];
147
+ if (primer) {
148
+ leading.push(primerMessage(primer));
149
+ }
150
+ if (summary) {
151
+ leading.push(summaryMessage(summary));
152
+ }
153
+ return [...leading, ...rawMessages];
154
+ }
155
+
156
+ return {
157
+ addUserMessage: (content) => add({ role: "user", content }),
158
+ addAssistantMessage: (content) => add({ role: "assistant", content }),
159
+ replaceLastAssistantMessage: (content) => {
160
+ const last = rawMessages[rawMessages.length - 1];
161
+ if (last?.role === "assistant") {
162
+ rawMessages[rawMessages.length - 1] = { role: "assistant", content };
163
+ }
164
+ },
165
+ getMessages,
166
+ getCharCount: () => getMessages().reduce((sum, m) => sum + m.content.length, 0),
167
+ };
168
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Confirm-required detection, re-exported from `@mercury-fw/channel-types`. The
3
+ * logic moved to the shared package so channel plugins can import it without
4
+ * depending on the app; kept re-exported here for the core callers
5
+ * (`agent-turn.ts` and the tests) that import it from this path.
6
+ */
7
+ export { detectPendingConfirmation } from "@mercury-fw/channel-types";
8
+ export type { PendingConfirmation } from "@mercury-fw/channel-types";
@@ -0,0 +1,38 @@
1
+ /**
2
+ * Model-invocable loader for a plugin's Agent Skill body — the "load" half of
3
+ * the discover-then-load shape SKILL.md introduces. The system prompt carries
4
+ * only each skill's name and one-line description (see `buildSystemPrompt`);
5
+ * when a request matches one, the model calls this to pull the skill's full
6
+ * instructions into context, exactly as it calls a CLI's `--help` rather than
7
+ * carrying every flag in the prompt. Same `tool()`-wrapping split as
8
+ * `tool-log-recall-tool.ts` and `wiki-tools.ts`.
9
+ *
10
+ * Registered by the composition root only when at least one skill loaded, so an
11
+ * instance with no skills never exposes an empty tool.
12
+ */
13
+ import { tool } from "ai";
14
+ import { z } from "zod";
15
+ import type { Skill, ExecutableTool } from "@mercury-fw/plugin-types";
16
+
17
+ export function createReadSkillTool(skills: Skill[]): { read_skill: ExecutableTool } {
18
+ const byName = new Map(skills.map((s) => [s.name, s]));
19
+ const names = skills.map((s) => s.name).join(", ");
20
+ const read_skill = tool({
21
+ description:
22
+ "Load the full instructions for one of your skills by name. Each skill's name and one-line description " +
23
+ "is listed in your system prompt; call this to get its detailed how-to (conventions, flags, examples) " +
24
+ `BEFORE acting on a request it covers — the description alone is not enough. Available skills: ${names || "(none)"}.`,
25
+ inputSchema: z.object({
26
+ name: z.string().describe("The exact skill name from the 'Available skills' list in the system prompt."),
27
+ }),
28
+ execute: async ({ name }) => {
29
+ const skill = byName.get(name);
30
+ if (!skill) {
31
+ return { ok: false as const, error: `no skill named "${name}". Available skills: ${names || "(none)"}.` };
32
+ }
33
+ return { ok: true as const, data: skill.body };
34
+ },
35
+ });
36
+
37
+ return { read_skill };
38
+ }
@@ -0,0 +1,69 @@
1
+ /**
2
+ * Turns a closed session's messages into structured `{topic, value}`
3
+ * facts for the semantic consolidation engine (see
4
+ * `src/memory/semantic-facts-store.ts`) — distinct from
5
+ * `episodic-summarizer.ts`, which produces a prose account of the whole
6
+ * session. A single extracted fact here is a candidate, not yet a
7
+ * standing belief about the user: consolidation (a separate,
8
+ * deterministic step) decides whether repeated facts on the same topic
9
+ * are frequent enough to be promoted to a wiki note.
10
+ */
11
+ import { generateObject, type LanguageModel } from "ai";
12
+ import { z } from "zod";
13
+ import type { Message } from "./history.ts";
14
+
15
+ /**
16
+ * Closed vocabulary — the model can only ever return one of these exact
17
+ * values, never invent a new key for the same concept. Deliberately
18
+ * excludes identity/name: a registered Chat app's own `MESSAGE` event
19
+ * already carries the sender's `displayName` directly, so a semantic
20
+ * fact about "who the user is" would only duplicate or contradict that
21
+ * more authoritative source, never add anything — observed live as the
22
+ * `name`/`user-name` duplicate before this fix.
23
+ */
24
+ export const SEMANTIC_FACT_TOPICS = ["team", "role", "preferred-language", "tools-used"] as const;
25
+ export const SemanticFactSchema = z.object({ topic: z.enum(SEMANTIC_FACT_TOPICS), value: z.string() });
26
+ export type SemanticFact = z.infer<typeof SemanticFactSchema>;
27
+
28
+ type GenerateObjectFn = (params: {
29
+ model: LanguageModel;
30
+ output: "array";
31
+ schema: typeof SemanticFactSchema;
32
+ instructions: string;
33
+ prompt: string;
34
+ }) => Promise<{ object: SemanticFact[] }>;
35
+
36
+ const SYSTEM_PROMPT =
37
+ "Estrai fatti stabili e ricorrenti sull'utente da questa conversazione, scegliendo il topic " +
38
+ 'esclusivamente tra questi quattro: "team", "role" (ruolo), "preferred-language" (lingua ' +
39
+ 'preferita), "tools-used" (strumenti usati). Ogni fatto è una coppia {topic, value}: "value" è ' +
40
+ "quanto dichiarato o chiaramente implicato per quel topic. Non estrarre l'identità o il nome " +
41
+ "dell'utente — Mercury lo traccia già separatamente, non va incluso qui. Non estrarre dettagli " +
42
+ "specifici di un singolo task, validi solo per questa sessione — solo cose plausibilmente vere " +
43
+ "anche in futuro. Restituisci un array vuoto se non c'è nulla che qualifica tra i topic ammessi.";
44
+
45
+ // index.ts prepends this to every user message before it reaches history,
46
+ // so the model knows who it's talking to within a turn — bookkeeping
47
+ // Mercury wrote itself, never something the user said. Left in, the
48
+ // extractor mistakes the marker repeating in every turn for a "stable,
49
+ // recurring fact" (this is exactly how the live name/user-name duplicate
50
+ // happened): stripped here, only in this extractor, before the messages
51
+ // are joined into the prompt.
52
+ const SENDER_MARKER_RE = /^\[Da: [^\]]*\]\n/;
53
+
54
+ /** Returns a function that extracts standing {topic, value} facts from a closed session's messages. */
55
+ export function createSemanticFactExtractor(
56
+ model: LanguageModel,
57
+ generateObjectFn: GenerateObjectFn = generateObject as unknown as GenerateObjectFn,
58
+ ): (messages: Message[]) => Promise<SemanticFact[]> {
59
+ return async (messages) => {
60
+ const { object } = await generateObjectFn({
61
+ model,
62
+ output: "array",
63
+ schema: SemanticFactSchema,
64
+ instructions: SYSTEM_PROMPT,
65
+ prompt: messages.map((m) => `${m.role}: ${m.content.replace(SENDER_MARKER_RE, "")}`).join("\n"),
66
+ });
67
+ return object;
68
+ };
69
+ }
@@ -0,0 +1,27 @@
1
+ /**
2
+ * Minimal shape of a finished generation step that callers might care
3
+ * about — just enough to show what tool Mercury called, with what
4
+ * input, and what it got back. `toolCallId` is what links an entry in
5
+ * `toolCalls` to its entry in `toolResults` — a call with no matching
6
+ * result is a real case callers need to handle explicitly rather than
7
+ * assume a 1:1 pairing: it means the call failed before ever executing
8
+ * (e.g. malformed arguments that don't match the tool's schema), which
9
+ * shows up as a `tool-error` entry in `content`, not in `toolResults` —
10
+ * confirmed against the real AI SDK's `StepResult` type, which has no
11
+ * separate `toolErrors` array; `content` is the one place every part
12
+ * (text/tool-call/tool-result/tool-error) actually lives. The real AI
13
+ * SDK step object has many more fields; this is a subset, which is fine
14
+ * since function parameter types only need to be structurally
15
+ * compatible, not identical.
16
+ *
17
+ * Lives in its own file, not `agent-turn.ts`, specifically so
18
+ * `pending-confirmation.ts` (which needs this type) and `agent-turn.ts`
19
+ * (which needs `pending-confirmation.ts`'s `detectPendingConfirmation` to
20
+ * decide whether to stop the tool-calling loop early) don't form an
21
+ * import cycle.
22
+ */
23
+ // `StepInfo` is part of the plugin contract (a `PostTurnGuard`'s `run`
24
+ // receives `StepInfo[]`), so it lives in `@mercury-fw/plugin-types` and is
25
+ // re-exported here for the many core callers that import it from this module.
26
+ import type { StepInfo } from "@mercury-fw/plugin-types";
27
+ export type { StepInfo };
@@ -0,0 +1,36 @@
1
+ /**
2
+ * Thin glue that turns `createSessionHistory`'s injected `summarize`
3
+ * dependency into a real LLM call.
4
+ *
5
+ * Kept separate from `history.ts` on purpose: `history.ts`'s threshold/
6
+ * merge/replace logic is the substantial part and is fully unit-tested
7
+ * with a fake summarizer; this file's only job is producing the prompt
8
+ * and calling `generateText`, which isn't worth mocking deeply for one
9
+ * line of glue.
10
+ *
11
+ * Used by: `src/index.ts` (wiring), which passes the result into
12
+ * `createSessionHistory` (see `src/session/history.ts`).
13
+ */
14
+ import { generateText, type LanguageModel } from "ai";
15
+ import type { Message } from "./history.ts";
16
+
17
+ /**
18
+ * Returns a function matching `createSessionHistory`'s `summarize`
19
+ * signature, backed by `model`. The returned function asks the model to
20
+ * condense the given messages into a short summary that preserves
21
+ * names, ticket keys, and decisions — the things a future turn would
22
+ * otherwise lose once the raw messages are cleared.
23
+ */
24
+ export function createSummarizer(
25
+ model: LanguageModel,
26
+ ): (messages: Message[]) => Promise<string> {
27
+ return async (messages) => {
28
+ const { text } = await generateText({
29
+ model,
30
+ instructions:
31
+ "Summarize this conversation concisely, preserving names, ticket keys, and decisions.",
32
+ prompt: messages.map((m) => `${m.role}: ${m.content}`).join("\n"),
33
+ });
34
+ return text;
35
+ };
36
+ }
@@ -0,0 +1,142 @@
1
+ import { NO_REPLY } from "@mercury-fw/channel-types";
2
+ import type { Skill } from "@mercury-fw/plugin-types";
3
+
4
+ /** The assistant's persona, set by the instance: `identity` opens the system
5
+ * prompt, `tone` closes it. Either one left out falls back to its default. */
6
+ export type Persona = { identity?: string; tone?: string };
7
+
8
+ /** The identity an instance gets when its config sets none. */
9
+ export const DEFAULT_PERSONA_IDENTITY = "You are Mercury, an internal assistant.";
10
+
11
+ /** The tone an instance gets when its config sets none. */
12
+ export const DEFAULT_PERSONA_TONE = [
13
+ "DO:",
14
+ "- Answer directly, in plain text only.",
15
+ "- Be dry but respectful, and complete.",
16
+ "- If you believe a point of view is useful, add it — but keep it brief and put it strictly at the end.",
17
+ "",
18
+ "DON'T:",
19
+ "- DON'T use Markdown formatting (no **, #, -, etc.), unless the user explicitly asks for it.",
20
+ "- DON'T introduce yourself as Mercury unless asked; the user already knows who you are.",
21
+ "- DON'T ask follow-up questions.",
22
+ "- DON'T add extra explanations or extra actions beyond what was requested.",
23
+ ].join("\n");
24
+
25
+ /** A persona field's text, trimmed at the end (one kept in a Markdown file and
26
+ * imported as text always ends with a newline); empty or whitespace-only counts
27
+ * as left out and yields the default. */
28
+ function personaSlot(text: string | undefined, fallback: string): string {
29
+ const trimmed = text?.trimEnd();
30
+ return trimmed ? trimmed : fallback;
31
+ }
32
+
33
+ /**
34
+ * Builds a system prompt that only describes tools actually present in
35
+ * `tools` (see `src/session/agent-turn.ts` for why a prompt mentioning
36
+ * an absent tool is a real bug, not a harmless no-op). The persona fills the
37
+ * first and last slots (see `personaSlot`).
38
+ */
39
+ export function buildSystemPrompt(opts: {
40
+ pluginFragments: string[];
41
+ skills: Skill[];
42
+ multiUserChannel: boolean;
43
+ persona?: Persona;
44
+ }): string {
45
+ const lines = [personaSlot(opts.persona?.identity, DEFAULT_PERSONA_IDENTITY)];
46
+ // Each loaded plugin's own always-on system-prompt fragment, inserted
47
+ // verbatim in the order the composition root supplies them. A plugin that
48
+ // failed to load contributes nothing — the fragment and the tool now come
49
+ // from the same place, which is what retires the "prompt describes a tool
50
+ // this instance doesn't have" bug class.
51
+ for (const fragment of opts.pluginFragments) {
52
+ lines.push(fragment);
53
+ }
54
+ // Skill descriptors (Agent Skills): only the name + one-line description stays
55
+ // in the prompt, always — enough to know a capability exists. The voluminous
56
+ // body loads on demand via the read_skill tool, the same discover-then-load
57
+ // shape the CLIs already use with --help, instead of paying for every skill's
58
+ // full instructions on every turn.
59
+ if (opts.skills.length > 0) {
60
+ lines.push(
61
+ [
62
+ "You have skills — capabilities whose detailed instructions you load on demand. Before acting on a request a skill covers, call read_skill with its name to load its full how-to first; the descriptions below say only which skill applies, never how to use it.",
63
+ "Available skills:",
64
+ ...opts.skills.map((s) => `- ${s.name}: ${s.description}`),
65
+ ].join("\n"),
66
+ );
67
+ }
68
+ // Always present (WIKI_VAULT_PATH is a required env var, the vault
69
+ // always exists once Mercury boots) — unlike jira, this
70
+ // block doesn't need its own opts flag.
71
+ lines.push(
72
+ [
73
+ "You have access to wiki tools: list_files, read_file, grep, write_file, resolve_reference — Mercury's own knowledge base. " +
74
+ "curated/ is team knowledge (conventions, docs, project status) — written by maintainers, and by you. " +
75
+ "inferred/ is private per-user notes managed automatically by a separate process, not by you directly.",
76
+ "DO:",
77
+ "- If your context contains an opaque `[REQ:<token>]` marker, that's a reference to a past confirm-required request — call resolve_reference with that token to see what it was, don't guess at what it means.",
78
+ "- For a CLI's own syntax/flags, rely on --help first. Only check the wiki if --help doesn't cover something specific to how this team uses that tool (a convention, a naming pattern, a policy).",
79
+ "- When a command's --select flag description is generic/shared across multiple subcommands, don't take its inline example at face value — check that command's own \"Examples\" section at the bottom of its --help output for the syntax that actually works with it.",
80
+ "- For anything else — documentation, project status, how some tool or process is used, team conventions — consult the wiki FIRST (grep/read_file/list_files), before trying a CLI or answering from general knowledge.",
81
+ "- If the wiki doesn't have the answer, try a live CLI query if one is relevant, before giving up.",
82
+ "- If you still don't know after checking both, say so plainly — don't guess or invent an answer.",
83
+ "- If you learn something worth remembering (a useful command pattern, a correction from the user, a new convention), write_file to add it to curated/ — prefer creating a new, clearly-named file over guessing at how to merge into an existing one.",
84
+ "",
85
+ "DON'T:",
86
+ "- DON'T claim something is documented in the wiki without actually reading it via read_file/grep first.",
87
+ "- DON'T write_file over an existing curated document without reading it first — write_file replaces the whole file, it doesn't merge, so an unread overwrite silently destroys whatever was already there.",
88
+ ].join("\n"),
89
+ );
90
+ lines.push(
91
+ [
92
+ "You have access to the recall_tool_calls tool.",
93
+ "DO:",
94
+ "- If asked what you actually ran/queried/did earlier in this same conversation, call recall_tool_calls and quote it verbatim — you have no memory of your own past tool calls otherwise, only your own prior reply text, so reconstructing from memory instead of calling this tool risks getting it wrong.",
95
+ ].join("\n"),
96
+ );
97
+ lines.push(
98
+ [
99
+ "You have access to the recall_verbatim tool.",
100
+ "DO:",
101
+ "- If asked about something said in an EARLIER conversation — beyond what you can see in this one — call recall_verbatim to retrieve the actual past messages and quote them, don't reconstruct from memory. This searches a durable archive of what you and this person really said before.",
102
+ "- Use recall_tool_calls, not this, for what you ran in the CURRENT conversation; use recall_verbatim for what was said in past ones.",
103
+ ].join("\n"),
104
+ );
105
+
106
+ if (opts.multiUserChannel) {
107
+ // Interim, explicitly non-deterministic mitigation for Mercury replying
108
+ // to every message in a shared space — not a replacement for real
109
+ // @-mention detection, which the registered app's own identity now
110
+ // makes possible but which isn't implemented yet. See NO_REPLY in
111
+ // google-chat-provider.ts for the code side of this check.
112
+ lines.push(
113
+ [
114
+ "This conversation may be a shared space with more than one person, not a private one-on-one chat.",
115
+ "DO:",
116
+ "- Only give a substantive answer if this message is clearly directed at you (e.g. it explicitly mentions/addresses you) or is a direct continuation of an exchange you were already having with this same sender.",
117
+ `- If the message doesn't seem directed at you or isn't relevant to you, respond with exactly \`${NO_REPLY}\` and nothing else — no punctuation, no explanation, nothing before or after it.`,
118
+ ].join("\n"),
119
+ );
120
+ }
121
+
122
+ lines.push(personaSlot(opts.persona?.tone, DEFAULT_PERSONA_TONE));
123
+ return lines.join("\n");
124
+ }
125
+
126
+ /**
127
+ * Builds the instance's two system prompts from the same plugins and persona:
128
+ * `system` for 1:1 channels (the terminal, the HTTP surface) and `chatSystem`
129
+ * for shared spaces, the only one that carries the multi-user clause — an
130
+ * operator typing normally must never get a NO_REPLY meant for a shared space.
131
+ */
132
+ export function buildSystemPrompts(opts: {
133
+ pluginFragments: string[];
134
+ skills: Skill[];
135
+ /** Required even when undefined, so a caller can't silently drop the instance's persona. */
136
+ persona: Persona | undefined;
137
+ }): { system: string; chatSystem: string } {
138
+ return {
139
+ system: buildSystemPrompt({ ...opts, multiUserChannel: false }),
140
+ chatSystem: buildSystemPrompt({ ...opts, multiUserChannel: true }),
141
+ };
142
+ }
@@ -0,0 +1,133 @@
1
+ /**
2
+ * Detects, within a single turn's steps, a CLI command call that failed
3
+ * followed later by one for the same binary that succeeded — a candidate
4
+ * procedural correction, distinct from `semantic-fact-extractor.ts` (which
5
+ * is about the user, from `session.messages`) — this is about a *tool*,
6
+ * from the tool-call trace (`StepInfo`), true for whoever uses that
7
+ * command next, not just this user. Pairing is deterministic (pure
8
+ * string/status matching); describing *what* the correction actually was
9
+ * is delegated to the model, same `generateObject`-with-fixed-schema style
10
+ * as `semantic-fact-extractor.ts`.
11
+ *
12
+ * Wired per-turn (`onStepFinish`, see `index.ts`), not from
13
+ * `idle-session-cron.ts`'s idle sweep: the tool-call trace for a turn only
14
+ * exists in memory for the duration of that turn (`tool-log-buffer.ts` is a
15
+ * 200-entry ring buffer shared across every session — not a reliable place
16
+ * to reconstruct one turn's trace minutes or hours later).
17
+ */
18
+ import { generateObject, type LanguageModel } from "ai";
19
+ import { z } from "zod";
20
+ import type { StepInfo } from "./step-info.ts";
21
+
22
+ export const ProceduralCorrectionCandidateSchema = z.object({ topic: z.string(), value: z.string() });
23
+ export type ProceduralCorrection = { tool: string; topic: string; value: string };
24
+
25
+ type GenerateObjectFn = (params: {
26
+ model: LanguageModel;
27
+ output: "array";
28
+ schema: typeof ProceduralCorrectionCandidateSchema;
29
+ instructions: string;
30
+ prompt: string;
31
+ }) => Promise<{ object: z.infer<typeof ProceduralCorrectionCandidateSchema>[] }>;
32
+
33
+ const SYSTEM_PROMPT =
34
+ "Ti viene mostrato un tentativo di comando fallito e uno, per lo stesso strumento, andato a buon " +
35
+ "fine più tardi nello stesso turno di conversazione. Descrivi la correzione appresa come una coppia " +
36
+ '{topic, value}: "topic" è una chiave breve e stabile per questa specifica correzione (es. ' +
37
+ '"select-prefix", "assignee-operator"), "value" è la regola pratica da ricordare, in una frase, utile ' +
38
+ "per chiunque userà questo comando in futuro — non solo per chi l'ha scoperta ora. Se il secondo " +
39
+ "tentativo non è davvero una correzione dell'errore del primo (es. l'utente ha semplicemente cambiato " +
40
+ "richiesta), restituisci un array vuoto.";
41
+
42
+ /** Same normalization `semantic-fact-extractor.ts` used to apply to identity/preference topics — still needed here since this topic is free text, not a closed enum (procedural corrections are open-ended by nature). */
43
+ function normalizeTopic(topic: string): string {
44
+ return topic
45
+ .trim()
46
+ .toLowerCase()
47
+ .replace(/[\s_]+/g, "-");
48
+ }
49
+
50
+ type Attempt = { command: string; binary: string; ok: boolean; error?: string };
51
+
52
+ function extractAttempts(steps: StepInfo[]): Attempt[] {
53
+ const attempts: Attempt[] = [];
54
+ for (const stepInfo of steps) {
55
+ for (const call of stepInfo.toolCalls) {
56
+ // A CLI attempt is any tool call carrying a `command` string — the
57
+ // generic shape every CLI tool takes, whatever its name (jiraCommand,
58
+ // bitbucketCommand, the residual runCommand). Keying on a fixed tool name
59
+ // would silently miss plugin-owned CLI tools; keying on the input shape
60
+ // stays correct as new CLI plugins are added.
61
+ const input = call.input as { command?: unknown };
62
+ if (typeof input.command !== "string") continue;
63
+ // Group attempts by the binary they invoked, which is just the command's
64
+ // first whitespace-delimited token — no need to fully tokenize the argv
65
+ // (that belongs to whoever executes the command, not to procedural
66
+ // learning). An empty/whitespace-only command has no binary to group by.
67
+ const binary = input.command.trim().split(/\s+/)[0] ?? "";
68
+ if (binary === "") continue;
69
+
70
+ const result = stepInfo.toolResults.find((r) => r.toolCallId === call.toolCallId);
71
+ const output = result?.output as { ok?: unknown; error?: unknown } | undefined;
72
+ if (!output) continue; // no result to learn from (e.g. malformed-args tool-error, not a CLI failure)
73
+
74
+ attempts.push({
75
+ command: input.command,
76
+ binary,
77
+ ok: output.ok === true,
78
+ error: typeof output.error === "string" ? output.error : undefined,
79
+ });
80
+ }
81
+ }
82
+ return attempts;
83
+ }
84
+
85
+ /** Each failed attempt paired with the first later attempt for the same binary that succeeded — never the reverse, and never across different binaries. */
86
+ function findCorrectionPairs(attempts: Attempt[]): Array<{ failed: Attempt; corrected: Attempt }> {
87
+ const pairs: Array<{ failed: Attempt; corrected: Attempt }> = [];
88
+ for (let i = 0; i < attempts.length; i++) {
89
+ const failed = attempts[i];
90
+ if (!failed || failed.ok) continue;
91
+ const corrected = attempts.slice(i + 1).find((a) => a.binary === failed.binary && a.ok);
92
+ if (corrected) {
93
+ pairs.push({ failed, corrected });
94
+ }
95
+ }
96
+ return pairs;
97
+ }
98
+
99
+ /** Returns a function that extracts candidate procedural corrections from a single turn's steps. */
100
+ export function createToolCorrectionExtractor(
101
+ model: LanguageModel,
102
+ generateObjectFn: GenerateObjectFn = generateObject as unknown as GenerateObjectFn,
103
+ deps?: { log?: (msg: string) => void },
104
+ ): (steps: StepInfo[]) => Promise<ProceduralCorrection[]> {
105
+ const log = deps?.log ?? ((msg: string) => console.error(msg));
106
+
107
+ return async (steps) => {
108
+ const pairs = findCorrectionPairs(extractAttempts(steps));
109
+ const corrections: ProceduralCorrection[] = [];
110
+
111
+ for (const { failed, corrected } of pairs) {
112
+ try {
113
+ const { object } = await generateObjectFn({
114
+ model,
115
+ output: "array",
116
+ schema: ProceduralCorrectionCandidateSchema,
117
+ instructions: SYSTEM_PROMPT,
118
+ prompt:
119
+ `Comando fallito: ${failed.command}\n` +
120
+ `Errore: ${failed.error ?? "(nessun messaggio)"}\n` +
121
+ `Comando corretto (riuscito): ${corrected.command}`,
122
+ });
123
+ for (const candidate of object) {
124
+ corrections.push({ tool: failed.binary, topic: normalizeTopic(candidate.topic), value: candidate.value });
125
+ }
126
+ } catch (err) {
127
+ log(`procedural correction extraction failed for ${failed.binary}: ${String(err)}`);
128
+ }
129
+ }
130
+
131
+ return corrections;
132
+ };
133
+ }