@mercury-fw/core 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +38 -0
- package/dist/index.d.ts +23 -0
- package/dist/src/admin/cli-routes.d.ts +22 -0
- package/dist/src/admin/env-file.d.ts +1 -0
- package/dist/src/admin/model-routes.d.ts +26 -0
- package/dist/src/admin/qdrant-scroll.d.ts +34 -0
- package/dist/src/admin/server.d.ts +40 -0
- package/dist/src/admin/wiki-routes.d.ts +31 -0
- package/dist/src/compose.d.ts +42 -0
- package/dist/src/config/define-config.d.ts +31 -0
- package/dist/src/cron/idle-session-cron.d.ts +80 -0
- package/dist/src/cron/idle-session-scanner.d.ts +16 -0
- package/dist/src/cron/self-review-cron.d.ts +55 -0
- package/dist/src/cron/semantic-consolidation.d.ts +71 -0
- package/dist/src/memory/embedder.d.ts +9 -0
- package/dist/src/memory/episodic-store.d.ts +121 -0
- package/dist/src/memory/memory-provider.d.ts +51 -0
- package/dist/src/memory/semantic-facts-store.d.ts +37 -0
- package/dist/src/memory/tool-corrections-store.d.ts +26 -0
- package/dist/src/memory/verbatim-archive-store.d.ts +86 -0
- package/dist/src/model/client.d.ts +24 -0
- package/dist/src/model/context-size.d.ts +30 -0
- package/dist/src/plugins/manifest.d.ts +29 -0
- package/dist/src/plugins/plugin-loader.d.ts +85 -0
- package/dist/src/router/channel-loader.d.ts +30 -0
- package/dist/src/router/provider.d.ts +7 -0
- package/dist/src/router/terminal-provider.d.ts +37 -0
- package/dist/src/router/terminal.d.ts +41 -0
- package/dist/src/router/tool-log.d.ts +65 -0
- package/dist/src/router/turn-runner.d.ts +86 -0
- package/dist/src/session/agent-turn.d.ts +266 -0
- package/dist/src/session/context-primer.d.ts +16 -0
- package/dist/src/session/episodic-summarizer.d.ts +25 -0
- package/dist/src/session/history.d.ts +95 -0
- package/dist/src/session/pending-confirmation.d.ts +8 -0
- package/dist/src/session/read-skill-tool.d.ts +4 -0
- package/dist/src/session/semantic-fact-extractor.d.ts +45 -0
- package/dist/src/session/step-info.d.ts +24 -0
- package/dist/src/session/summarizer.d.ts +23 -0
- package/dist/src/session/system-prompt.d.ts +38 -0
- package/dist/src/session/tool-correction-extractor.d.ts +43 -0
- package/dist/src/session/tool-log-buffer.d.ts +24 -0
- package/dist/src/session/tool-log-recall-tool.d.ts +18 -0
- package/dist/src/session/tool-start-hook.d.ts +57 -0
- package/dist/src/tools/display-store.d.ts +36 -0
- package/dist/src/tools/present-tool.d.ts +23 -0
- package/dist/src/wiki/frontmatter-schema.d.ts +53 -0
- package/dist/src/wiki/index-entry.d.ts +15 -0
- package/dist/src/wiki/orphan-detector.d.ts +1 -0
- package/dist/src/wiki/self-review-runner.d.ts +48 -0
- package/dist/src/wiki/self-review-tools.d.ts +22 -0
- package/dist/src/wiki/vault-cli.d.ts +2 -0
- package/dist/src/wiki/vault-init.d.ts +7 -0
- package/dist/src/wiki/wiki-note.d.ts +62 -0
- package/dist/src/wiki/wiki-read.d.ts +27 -0
- package/dist/src/wiki/wiki-tools.d.ts +7 -0
- package/index.ts +23 -0
- package/package.json +49 -0
- package/src/admin/cli-routes.ts +48 -0
- package/src/admin/env-file.ts +29 -0
- package/src/admin/model-routes.ts +71 -0
- package/src/admin/public/index.html +416 -0
- package/src/admin/qdrant-scroll.ts +45 -0
- package/src/admin/server.ts +188 -0
- package/src/admin/wiki-routes.ts +93 -0
- package/src/compose.ts +599 -0
- package/src/config/define-config.ts +35 -0
- package/src/cron/.gitkeep +0 -0
- package/src/cron/idle-session-cron.ts +144 -0
- package/src/cron/idle-session-scanner.ts +37 -0
- package/src/cron/self-review-cron.ts +103 -0
- package/src/cron/semantic-consolidation.ts +228 -0
- package/src/memory/.gitkeep +0 -0
- package/src/memory/embedder.ts +15 -0
- package/src/memory/episodic-store.ts +183 -0
- package/src/memory/memory-provider.ts +98 -0
- package/src/memory/semantic-facts-store.ts +89 -0
- package/src/memory/tool-corrections-store.ts +72 -0
- package/src/memory/verbatim-archive-store.ts +202 -0
- package/src/model/client.ts +33 -0
- package/src/model/context-size.ts +42 -0
- package/src/plugins/manifest.ts +47 -0
- package/src/plugins/plugin-loader.ts +205 -0
- package/src/router/channel-loader.ts +56 -0
- package/src/router/provider.ts +7 -0
- package/src/router/terminal-provider.ts +155 -0
- package/src/router/terminal.ts +151 -0
- package/src/router/tool-log.ts +116 -0
- package/src/router/turn-runner.ts +205 -0
- package/src/session/agent-turn.ts +391 -0
- package/src/session/context-primer.ts +134 -0
- package/src/session/episodic-summarizer.ts +38 -0
- package/src/session/history.ts +168 -0
- package/src/session/pending-confirmation.ts +8 -0
- package/src/session/read-skill-tool.ts +38 -0
- package/src/session/semantic-fact-extractor.ts +69 -0
- package/src/session/step-info.ts +27 -0
- package/src/session/summarizer.ts +36 -0
- package/src/session/system-prompt.ts +142 -0
- package/src/session/tool-correction-extractor.ts +133 -0
- package/src/session/tool-log-buffer.ts +73 -0
- package/src/session/tool-log-recall-tool.ts +38 -0
- package/src/session/tool-start-hook.ts +164 -0
- package/src/tools/display-store.ts +89 -0
- package/src/tools/present-tool.ts +41 -0
- package/src/wiki/.gitkeep +0 -0
- package/src/wiki/frontmatter-schema.ts +49 -0
- package/src/wiki/index-entry.ts +59 -0
- package/src/wiki/orphan-detector.ts +61 -0
- package/src/wiki/self-review-runner.ts +133 -0
- package/src/wiki/self-review-tools.ts +162 -0
- package/src/wiki/vault-cli.ts +143 -0
- package/src/wiki/vault-init.ts +43 -0
- package/src/wiki/wiki-note.ts +326 -0
- package/src/wiki/wiki-read.ts +122 -0
- package/src/wiki/wiki-tools.ts +112 -0
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Layer 1 conversation memory: a sliding window of raw messages that
|
|
3
|
+
* summarizes itself once it grows too large, instead of growing
|
|
4
|
+
* unbounded across a long conversation.
|
|
5
|
+
*
|
|
6
|
+
* Why it exists: without a bound, a multi-turn conversation eventually
|
|
7
|
+
* overflows the model's context window. This is the only memory layer
|
|
8
|
+
* Mercury has in M1 — Layer 2 (wiki) and Layer 3 (episodic/Qdrant) are
|
|
9
|
+
* later milestones, pure enrichment that the system must work without.
|
|
10
|
+
*
|
|
11
|
+
* Used by: `src/session/agent-turn.ts` (`runTurn`), which appends each
|
|
12
|
+
* turn's user/assistant messages here and reads `getMessages()` to build
|
|
13
|
+
* the prompt for the next generation call. `src/session/summarizer.ts`
|
|
14
|
+
* supplies the `summarize` function injected into `createSessionHistory`.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/** A single turn's worth of conversation content. */
|
|
18
|
+
export type Message = { role: "user" | "assistant"; content: string };
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Character-count threshold (not a real tokenizer count — a `chars/4`
|
|
22
|
+
* estimate is close enough given the wide margin in the model's context
|
|
23
|
+
* budget) above which the raw message history gets summarized and
|
|
24
|
+
* replaced. Exported so tests can construct fixtures that land exactly
|
|
25
|
+
* at, or just past, the boundary.
|
|
26
|
+
*/
|
|
27
|
+
export const MAX_HISTORY_CHARS = 60_000;
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Mutable conversation history for a single ongoing conversation
|
|
31
|
+
* (one per channel/space — see `src/index.ts`, never shared across
|
|
32
|
+
* conversations).
|
|
33
|
+
*/
|
|
34
|
+
export type SessionHistory = {
|
|
35
|
+
/** Appends a user turn, summarizing first if this push crosses the threshold. */
|
|
36
|
+
addUserMessage(content: string): Promise<void>;
|
|
37
|
+
/** Appends an assistant turn, summarizing first if this push crosses the threshold. */
|
|
38
|
+
addAssistantMessage(content: string): Promise<void>;
|
|
39
|
+
/**
|
|
40
|
+
* Overwrites the most recently added message with `content`, in place,
|
|
41
|
+
* if it exists and is an assistant message — used when a turn's
|
|
42
|
+
* assistant text is corrected after already being recorded (see
|
|
43
|
+
* turn-runner.ts's issue-list correction). No-ops if there is no last
|
|
44
|
+
* message, or if it isn't an assistant message (defensive; shouldn't
|
|
45
|
+
* happen given how this is called). Deliberately synchronous and skips
|
|
46
|
+
* the summarization threshold check entirely: this replaces content
|
|
47
|
+
* already counted by the original `addAssistantMessage` call, it isn't
|
|
48
|
+
* new content being added.
|
|
49
|
+
*/
|
|
50
|
+
replaceLastAssistantMessage(content: string): void;
|
|
51
|
+
/**
|
|
52
|
+
* The messages to feed into the next model call: the current summary
|
|
53
|
+
* (if one exists, as a synthetic leading message) followed by the raw
|
|
54
|
+
* messages accumulated since the last summarization.
|
|
55
|
+
*/
|
|
56
|
+
getMessages(): Message[];
|
|
57
|
+
/**
|
|
58
|
+
* Total character length of what `getMessages()` would currently
|
|
59
|
+
* return — a live read on how close this conversation is to
|
|
60
|
+
* `MAX_HISTORY_CHARS` (and so to triggering summarization). Exposed so
|
|
61
|
+
* a channel can show this to a human, e.g. to tell apart "the model
|
|
62
|
+
* lost track of something" from "the context is actually near full".
|
|
63
|
+
*/
|
|
64
|
+
getCharCount(): number;
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
/** Wraps a summary string as the synthetic leading message `getMessages()` prepends. */
|
|
68
|
+
function summaryMessage(summary: string): Message {
|
|
69
|
+
return { role: "assistant", content: `Earlier conversation summary: ${summary}` };
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Wraps a primer string as the synthetic leading message `getMessages()`
|
|
74
|
+
* prepends ahead of any `summary` message. Distinct wording from
|
|
75
|
+
* `summaryMessage` on purpose — the primer describes the user's last
|
|
76
|
+
* *closed* session, not a summary of *this* one, and must not be confused
|
|
77
|
+
* with it.
|
|
78
|
+
*/
|
|
79
|
+
function primerMessage(primer: string): Message {
|
|
80
|
+
return { role: "assistant", content: `Context from your last session: ${primer}` };
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Creates an empty `SessionHistory`.
|
|
85
|
+
*
|
|
86
|
+
* @param summarize - Called with the messages that precede the current
|
|
87
|
+
* user turn (any prior summary re-injected as a leading message) whenever
|
|
88
|
+
* a single append pushes the total content length over
|
|
89
|
+
* `MAX_HISTORY_CHARS`. Its return value becomes the new summary; the
|
|
90
|
+
* trailing run from the last user message onward is retained as the raw
|
|
91
|
+
* window rather than cleared, so the current turn is never summarized away
|
|
92
|
+
* (a model call always follows `addUserMessage`, and the primer/summary
|
|
93
|
+
* leading messages are both `role:"assistant"` — folding the user turn
|
|
94
|
+
* into them would hand the model a user-less array). The threshold check
|
|
95
|
+
* runs after every individual append (not once per turn), so the crossing
|
|
96
|
+
* point is caught precisely regardless of whether it's the user or
|
|
97
|
+
* assistant message that tips it over. A lone crossing user message with
|
|
98
|
+
* nothing before it is left live and `summarize` is not called.
|
|
99
|
+
* @param onBeforeCompress - Optional, called synchronously with the exact
|
|
100
|
+
* same batch `summarize` is about to receive, right before it's
|
|
101
|
+
* compressed out of the live context — a second, independent signal a
|
|
102
|
+
* caller can mirror to somewhere durable (see `idle-session-cron.ts`'s
|
|
103
|
+
* shared capture function) before that content stops being directly
|
|
104
|
+
* visible to the model. Fire-and-forget on purpose: this function must
|
|
105
|
+
* never block or fail Layer 1's own compression on an external write.
|
|
106
|
+
* @param primer - Optional, set once at creation from the user's last closed
|
|
107
|
+
* session (see `context-primer.ts`). Held as its own state, entirely
|
|
108
|
+
* independent from `summary`: it's never included in the batch passed to
|
|
109
|
+
* `summarize`, so a real compression event can't paraphrase or drop it.
|
|
110
|
+
* Leads `getMessages()` for the whole life of this history.
|
|
111
|
+
*/
|
|
112
|
+
export function createSessionHistory(
|
|
113
|
+
summarize: (messages: Message[]) => Promise<string>,
|
|
114
|
+
onBeforeCompress?: (messages: Message[]) => void,
|
|
115
|
+
primer?: string,
|
|
116
|
+
): SessionHistory {
|
|
117
|
+
let rawMessages: Message[] = [];
|
|
118
|
+
let summary: string | null = null;
|
|
119
|
+
|
|
120
|
+
async function add(message: Message): Promise<void> {
|
|
121
|
+
rawMessages.push(message);
|
|
122
|
+
|
|
123
|
+
const total = rawMessages.reduce((sum, m) => sum + m.content.length, 0);
|
|
124
|
+
if (total > MAX_HISTORY_CHARS) {
|
|
125
|
+
// Retain the trailing run from the last user message onward instead of
|
|
126
|
+
// clearing everything: getMessages() for a model call always follows an
|
|
127
|
+
// addUserMessage, and the primer/summary leading messages are both
|
|
128
|
+
// role:"assistant", so summarizing the current user turn away would hand
|
|
129
|
+
// Ollama a user-less array ("no user query found in messages") — bug #19.
|
|
130
|
+
// Compress only what precedes that turn; if nothing precedes it (a lone
|
|
131
|
+
// oversized user message), leave it live — its question can't be summarized.
|
|
132
|
+
const lastUserIdx = rawMessages.findLastIndex((m) => m.role === "user");
|
|
133
|
+
const toCompress = lastUserIdx >= 0 ? rawMessages.slice(0, lastUserIdx) : rawMessages;
|
|
134
|
+
const retained = lastUserIdx >= 0 ? rawMessages.slice(lastUserIdx) : [];
|
|
135
|
+
if (toCompress.length === 0) {
|
|
136
|
+
return;
|
|
137
|
+
}
|
|
138
|
+
const batch = summary ? [summaryMessage(summary), ...toCompress] : toCompress;
|
|
139
|
+
onBeforeCompress?.(batch);
|
|
140
|
+
summary = await summarize(batch);
|
|
141
|
+
rawMessages = retained;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function getMessages(): Message[] {
|
|
146
|
+
const leading: Message[] = [];
|
|
147
|
+
if (primer) {
|
|
148
|
+
leading.push(primerMessage(primer));
|
|
149
|
+
}
|
|
150
|
+
if (summary) {
|
|
151
|
+
leading.push(summaryMessage(summary));
|
|
152
|
+
}
|
|
153
|
+
return [...leading, ...rawMessages];
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
return {
|
|
157
|
+
addUserMessage: (content) => add({ role: "user", content }),
|
|
158
|
+
addAssistantMessage: (content) => add({ role: "assistant", content }),
|
|
159
|
+
replaceLastAssistantMessage: (content) => {
|
|
160
|
+
const last = rawMessages[rawMessages.length - 1];
|
|
161
|
+
if (last?.role === "assistant") {
|
|
162
|
+
rawMessages[rawMessages.length - 1] = { role: "assistant", content };
|
|
163
|
+
}
|
|
164
|
+
},
|
|
165
|
+
getMessages,
|
|
166
|
+
getCharCount: () => getMessages().reduce((sum, m) => sum + m.content.length, 0),
|
|
167
|
+
};
|
|
168
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Confirm-required detection, re-exported from `@mercury-fw/channel-types`. The
|
|
3
|
+
* logic moved to the shared package so channel plugins can import it without
|
|
4
|
+
* depending on the app; kept re-exported here for the core callers
|
|
5
|
+
* (`agent-turn.ts` and the tests) that import it from this path.
|
|
6
|
+
*/
|
|
7
|
+
export { detectPendingConfirmation } from "@mercury-fw/channel-types";
|
|
8
|
+
export type { PendingConfirmation } from "@mercury-fw/channel-types";
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Model-invocable loader for a plugin's Agent Skill body — the "load" half of
|
|
3
|
+
* the discover-then-load shape SKILL.md introduces. The system prompt carries
|
|
4
|
+
* only each skill's name and one-line description (see `buildSystemPrompt`);
|
|
5
|
+
* when a request matches one, the model calls this to pull the skill's full
|
|
6
|
+
* instructions into context, exactly as it calls a CLI's `--help` rather than
|
|
7
|
+
* carrying every flag in the prompt. Same `tool()`-wrapping split as
|
|
8
|
+
* `tool-log-recall-tool.ts` and `wiki-tools.ts`.
|
|
9
|
+
*
|
|
10
|
+
* Registered by the composition root only when at least one skill loaded, so an
|
|
11
|
+
* instance with no skills never exposes an empty tool.
|
|
12
|
+
*/
|
|
13
|
+
import { tool } from "ai";
|
|
14
|
+
import { z } from "zod";
|
|
15
|
+
import type { Skill, ExecutableTool } from "@mercury-fw/plugin-types";
|
|
16
|
+
|
|
17
|
+
export function createReadSkillTool(skills: Skill[]): { read_skill: ExecutableTool } {
|
|
18
|
+
const byName = new Map(skills.map((s) => [s.name, s]));
|
|
19
|
+
const names = skills.map((s) => s.name).join(", ");
|
|
20
|
+
const read_skill = tool({
|
|
21
|
+
description:
|
|
22
|
+
"Load the full instructions for one of your skills by name. Each skill's name and one-line description " +
|
|
23
|
+
"is listed in your system prompt; call this to get its detailed how-to (conventions, flags, examples) " +
|
|
24
|
+
`BEFORE acting on a request it covers — the description alone is not enough. Available skills: ${names || "(none)"}.`,
|
|
25
|
+
inputSchema: z.object({
|
|
26
|
+
name: z.string().describe("The exact skill name from the 'Available skills' list in the system prompt."),
|
|
27
|
+
}),
|
|
28
|
+
execute: async ({ name }) => {
|
|
29
|
+
const skill = byName.get(name);
|
|
30
|
+
if (!skill) {
|
|
31
|
+
return { ok: false as const, error: `no skill named "${name}". Available skills: ${names || "(none)"}.` };
|
|
32
|
+
}
|
|
33
|
+
return { ok: true as const, data: skill.body };
|
|
34
|
+
},
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
return { read_skill };
|
|
38
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Turns a closed session's messages into structured `{topic, value}`
|
|
3
|
+
* facts for the semantic consolidation engine (see
|
|
4
|
+
* `src/memory/semantic-facts-store.ts`) — distinct from
|
|
5
|
+
* `episodic-summarizer.ts`, which produces a prose account of the whole
|
|
6
|
+
* session. A single extracted fact here is a candidate, not yet a
|
|
7
|
+
* standing belief about the user: consolidation (a separate,
|
|
8
|
+
* deterministic step) decides whether repeated facts on the same topic
|
|
9
|
+
* are frequent enough to be promoted to a wiki note.
|
|
10
|
+
*/
|
|
11
|
+
import { generateObject, type LanguageModel } from "ai";
|
|
12
|
+
import { z } from "zod";
|
|
13
|
+
import type { Message } from "./history.ts";
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Closed vocabulary — the model can only ever return one of these exact
|
|
17
|
+
* values, never invent a new key for the same concept. Deliberately
|
|
18
|
+
* excludes identity/name: a registered Chat app's own `MESSAGE` event
|
|
19
|
+
* already carries the sender's `displayName` directly, so a semantic
|
|
20
|
+
* fact about "who the user is" would only duplicate or contradict that
|
|
21
|
+
* more authoritative source, never add anything — observed live as the
|
|
22
|
+
* `name`/`user-name` duplicate before this fix.
|
|
23
|
+
*/
|
|
24
|
+
export const SEMANTIC_FACT_TOPICS = ["team", "role", "preferred-language", "tools-used"] as const;
|
|
25
|
+
export const SemanticFactSchema = z.object({ topic: z.enum(SEMANTIC_FACT_TOPICS), value: z.string() });
|
|
26
|
+
export type SemanticFact = z.infer<typeof SemanticFactSchema>;
|
|
27
|
+
|
|
28
|
+
type GenerateObjectFn = (params: {
|
|
29
|
+
model: LanguageModel;
|
|
30
|
+
output: "array";
|
|
31
|
+
schema: typeof SemanticFactSchema;
|
|
32
|
+
instructions: string;
|
|
33
|
+
prompt: string;
|
|
34
|
+
}) => Promise<{ object: SemanticFact[] }>;
|
|
35
|
+
|
|
36
|
+
const SYSTEM_PROMPT =
|
|
37
|
+
"Estrai fatti stabili e ricorrenti sull'utente da questa conversazione, scegliendo il topic " +
|
|
38
|
+
'esclusivamente tra questi quattro: "team", "role" (ruolo), "preferred-language" (lingua ' +
|
|
39
|
+
'preferita), "tools-used" (strumenti usati). Ogni fatto è una coppia {topic, value}: "value" è ' +
|
|
40
|
+
"quanto dichiarato o chiaramente implicato per quel topic. Non estrarre l'identità o il nome " +
|
|
41
|
+
"dell'utente — Mercury lo traccia già separatamente, non va incluso qui. Non estrarre dettagli " +
|
|
42
|
+
"specifici di un singolo task, validi solo per questa sessione — solo cose plausibilmente vere " +
|
|
43
|
+
"anche in futuro. Restituisci un array vuoto se non c'è nulla che qualifica tra i topic ammessi.";
|
|
44
|
+
|
|
45
|
+
// index.ts prepends this to every user message before it reaches history,
|
|
46
|
+
// so the model knows who it's talking to within a turn — bookkeeping
|
|
47
|
+
// Mercury wrote itself, never something the user said. Left in, the
|
|
48
|
+
// extractor mistakes the marker repeating in every turn for a "stable,
|
|
49
|
+
// recurring fact" (this is exactly how the live name/user-name duplicate
|
|
50
|
+
// happened): stripped here, only in this extractor, before the messages
|
|
51
|
+
// are joined into the prompt.
|
|
52
|
+
const SENDER_MARKER_RE = /^\[Da: [^\]]*\]\n/;
|
|
53
|
+
|
|
54
|
+
/** Returns a function that extracts standing {topic, value} facts from a closed session's messages. */
|
|
55
|
+
export function createSemanticFactExtractor(
|
|
56
|
+
model: LanguageModel,
|
|
57
|
+
generateObjectFn: GenerateObjectFn = generateObject as unknown as GenerateObjectFn,
|
|
58
|
+
): (messages: Message[]) => Promise<SemanticFact[]> {
|
|
59
|
+
return async (messages) => {
|
|
60
|
+
const { object } = await generateObjectFn({
|
|
61
|
+
model,
|
|
62
|
+
output: "array",
|
|
63
|
+
schema: SemanticFactSchema,
|
|
64
|
+
instructions: SYSTEM_PROMPT,
|
|
65
|
+
prompt: messages.map((m) => `${m.role}: ${m.content.replace(SENDER_MARKER_RE, "")}`).join("\n"),
|
|
66
|
+
});
|
|
67
|
+
return object;
|
|
68
|
+
};
|
|
69
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Minimal shape of a finished generation step that callers might care
|
|
3
|
+
* about — just enough to show what tool Mercury called, with what
|
|
4
|
+
* input, and what it got back. `toolCallId` is what links an entry in
|
|
5
|
+
* `toolCalls` to its entry in `toolResults` — a call with no matching
|
|
6
|
+
* result is a real case callers need to handle explicitly rather than
|
|
7
|
+
* assume a 1:1 pairing: it means the call failed before ever executing
|
|
8
|
+
* (e.g. malformed arguments that don't match the tool's schema), which
|
|
9
|
+
* shows up as a `tool-error` entry in `content`, not in `toolResults` —
|
|
10
|
+
* confirmed against the real AI SDK's `StepResult` type, which has no
|
|
11
|
+
* separate `toolErrors` array; `content` is the one place every part
|
|
12
|
+
* (text/tool-call/tool-result/tool-error) actually lives. The real AI
|
|
13
|
+
* SDK step object has many more fields; this is a subset, which is fine
|
|
14
|
+
* since function parameter types only need to be structurally
|
|
15
|
+
* compatible, not identical.
|
|
16
|
+
*
|
|
17
|
+
* Lives in its own file, not `agent-turn.ts`, specifically so
|
|
18
|
+
* `pending-confirmation.ts` (which needs this type) and `agent-turn.ts`
|
|
19
|
+
* (which needs `pending-confirmation.ts`'s `detectPendingConfirmation` to
|
|
20
|
+
* decide whether to stop the tool-calling loop early) don't form an
|
|
21
|
+
* import cycle.
|
|
22
|
+
*/
|
|
23
|
+
// `StepInfo` is part of the plugin contract (a `PostTurnGuard`'s `run`
|
|
24
|
+
// receives `StepInfo[]`), so it lives in `@mercury-fw/plugin-types` and is
|
|
25
|
+
// re-exported here for the many core callers that import it from this module.
|
|
26
|
+
import type { StepInfo } from "@mercury-fw/plugin-types";
|
|
27
|
+
export type { StepInfo };
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Thin glue that turns `createSessionHistory`'s injected `summarize`
|
|
3
|
+
* dependency into a real LLM call.
|
|
4
|
+
*
|
|
5
|
+
* Kept separate from `history.ts` on purpose: `history.ts`'s threshold/
|
|
6
|
+
* merge/replace logic is the substantial part and is fully unit-tested
|
|
7
|
+
* with a fake summarizer; this file's only job is producing the prompt
|
|
8
|
+
* and calling `generateText`, which isn't worth mocking deeply for one
|
|
9
|
+
* line of glue.
|
|
10
|
+
*
|
|
11
|
+
* Used by: `src/index.ts` (wiring), which passes the result into
|
|
12
|
+
* `createSessionHistory` (see `src/session/history.ts`).
|
|
13
|
+
*/
|
|
14
|
+
import { generateText, type LanguageModel } from "ai";
|
|
15
|
+
import type { Message } from "./history.ts";
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Returns a function matching `createSessionHistory`'s `summarize`
|
|
19
|
+
* signature, backed by `model`. The returned function asks the model to
|
|
20
|
+
* condense the given messages into a short summary that preserves
|
|
21
|
+
* names, ticket keys, and decisions — the things a future turn would
|
|
22
|
+
* otherwise lose once the raw messages are cleared.
|
|
23
|
+
*/
|
|
24
|
+
export function createSummarizer(
|
|
25
|
+
model: LanguageModel,
|
|
26
|
+
): (messages: Message[]) => Promise<string> {
|
|
27
|
+
return async (messages) => {
|
|
28
|
+
const { text } = await generateText({
|
|
29
|
+
model,
|
|
30
|
+
instructions:
|
|
31
|
+
"Summarize this conversation concisely, preserving names, ticket keys, and decisions.",
|
|
32
|
+
prompt: messages.map((m) => `${m.role}: ${m.content}`).join("\n"),
|
|
33
|
+
});
|
|
34
|
+
return text;
|
|
35
|
+
};
|
|
36
|
+
}
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
import { NO_REPLY } from "@mercury-fw/channel-types";
|
|
2
|
+
import type { Skill } from "@mercury-fw/plugin-types";
|
|
3
|
+
|
|
4
|
+
/** The assistant's persona, set by the instance: `identity` opens the system
|
|
5
|
+
* prompt, `tone` closes it. Either one left out falls back to its default. */
|
|
6
|
+
export type Persona = { identity?: string; tone?: string };
|
|
7
|
+
|
|
8
|
+
/** The identity an instance gets when its config sets none. */
|
|
9
|
+
export const DEFAULT_PERSONA_IDENTITY = "You are Mercury, an internal assistant.";
|
|
10
|
+
|
|
11
|
+
/** The tone an instance gets when its config sets none. */
|
|
12
|
+
export const DEFAULT_PERSONA_TONE = [
|
|
13
|
+
"DO:",
|
|
14
|
+
"- Answer directly, in plain text only.",
|
|
15
|
+
"- Be dry but respectful, and complete.",
|
|
16
|
+
"- If you believe a point of view is useful, add it — but keep it brief and put it strictly at the end.",
|
|
17
|
+
"",
|
|
18
|
+
"DON'T:",
|
|
19
|
+
"- DON'T use Markdown formatting (no **, #, -, etc.), unless the user explicitly asks for it.",
|
|
20
|
+
"- DON'T introduce yourself as Mercury unless asked; the user already knows who you are.",
|
|
21
|
+
"- DON'T ask follow-up questions.",
|
|
22
|
+
"- DON'T add extra explanations or extra actions beyond what was requested.",
|
|
23
|
+
].join("\n");
|
|
24
|
+
|
|
25
|
+
/** A persona field's text, trimmed at the end (one kept in a Markdown file and
|
|
26
|
+
* imported as text always ends with a newline); empty or whitespace-only counts
|
|
27
|
+
* as left out and yields the default. */
|
|
28
|
+
function personaSlot(text: string | undefined, fallback: string): string {
|
|
29
|
+
const trimmed = text?.trimEnd();
|
|
30
|
+
return trimmed ? trimmed : fallback;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Builds a system prompt that only describes tools actually present in
|
|
35
|
+
* `tools` (see `src/session/agent-turn.ts` for why a prompt mentioning
|
|
36
|
+
* an absent tool is a real bug, not a harmless no-op). The persona fills the
|
|
37
|
+
* first and last slots (see `personaSlot`).
|
|
38
|
+
*/
|
|
39
|
+
export function buildSystemPrompt(opts: {
|
|
40
|
+
pluginFragments: string[];
|
|
41
|
+
skills: Skill[];
|
|
42
|
+
multiUserChannel: boolean;
|
|
43
|
+
persona?: Persona;
|
|
44
|
+
}): string {
|
|
45
|
+
const lines = [personaSlot(opts.persona?.identity, DEFAULT_PERSONA_IDENTITY)];
|
|
46
|
+
// Each loaded plugin's own always-on system-prompt fragment, inserted
|
|
47
|
+
// verbatim in the order the composition root supplies them. A plugin that
|
|
48
|
+
// failed to load contributes nothing — the fragment and the tool now come
|
|
49
|
+
// from the same place, which is what retires the "prompt describes a tool
|
|
50
|
+
// this instance doesn't have" bug class.
|
|
51
|
+
for (const fragment of opts.pluginFragments) {
|
|
52
|
+
lines.push(fragment);
|
|
53
|
+
}
|
|
54
|
+
// Skill descriptors (Agent Skills): only the name + one-line description stays
|
|
55
|
+
// in the prompt, always — enough to know a capability exists. The voluminous
|
|
56
|
+
// body loads on demand via the read_skill tool, the same discover-then-load
|
|
57
|
+
// shape the CLIs already use with --help, instead of paying for every skill's
|
|
58
|
+
// full instructions on every turn.
|
|
59
|
+
if (opts.skills.length > 0) {
|
|
60
|
+
lines.push(
|
|
61
|
+
[
|
|
62
|
+
"You have skills — capabilities whose detailed instructions you load on demand. Before acting on a request a skill covers, call read_skill with its name to load its full how-to first; the descriptions below say only which skill applies, never how to use it.",
|
|
63
|
+
"Available skills:",
|
|
64
|
+
...opts.skills.map((s) => `- ${s.name}: ${s.description}`),
|
|
65
|
+
].join("\n"),
|
|
66
|
+
);
|
|
67
|
+
}
|
|
68
|
+
// Always present (WIKI_VAULT_PATH is a required env var, the vault
|
|
69
|
+
// always exists once Mercury boots) — unlike jira, this
|
|
70
|
+
// block doesn't need its own opts flag.
|
|
71
|
+
lines.push(
|
|
72
|
+
[
|
|
73
|
+
"You have access to wiki tools: list_files, read_file, grep, write_file, resolve_reference — Mercury's own knowledge base. " +
|
|
74
|
+
"curated/ is team knowledge (conventions, docs, project status) — written by maintainers, and by you. " +
|
|
75
|
+
"inferred/ is private per-user notes managed automatically by a separate process, not by you directly.",
|
|
76
|
+
"DO:",
|
|
77
|
+
"- If your context contains an opaque `[REQ:<token>]` marker, that's a reference to a past confirm-required request — call resolve_reference with that token to see what it was, don't guess at what it means.",
|
|
78
|
+
"- For a CLI's own syntax/flags, rely on --help first. Only check the wiki if --help doesn't cover something specific to how this team uses that tool (a convention, a naming pattern, a policy).",
|
|
79
|
+
"- When a command's --select flag description is generic/shared across multiple subcommands, don't take its inline example at face value — check that command's own \"Examples\" section at the bottom of its --help output for the syntax that actually works with it.",
|
|
80
|
+
"- For anything else — documentation, project status, how some tool or process is used, team conventions — consult the wiki FIRST (grep/read_file/list_files), before trying a CLI or answering from general knowledge.",
|
|
81
|
+
"- If the wiki doesn't have the answer, try a live CLI query if one is relevant, before giving up.",
|
|
82
|
+
"- If you still don't know after checking both, say so plainly — don't guess or invent an answer.",
|
|
83
|
+
"- If you learn something worth remembering (a useful command pattern, a correction from the user, a new convention), write_file to add it to curated/ — prefer creating a new, clearly-named file over guessing at how to merge into an existing one.",
|
|
84
|
+
"",
|
|
85
|
+
"DON'T:",
|
|
86
|
+
"- DON'T claim something is documented in the wiki without actually reading it via read_file/grep first.",
|
|
87
|
+
"- DON'T write_file over an existing curated document without reading it first — write_file replaces the whole file, it doesn't merge, so an unread overwrite silently destroys whatever was already there.",
|
|
88
|
+
].join("\n"),
|
|
89
|
+
);
|
|
90
|
+
lines.push(
|
|
91
|
+
[
|
|
92
|
+
"You have access to the recall_tool_calls tool.",
|
|
93
|
+
"DO:",
|
|
94
|
+
"- If asked what you actually ran/queried/did earlier in this same conversation, call recall_tool_calls and quote it verbatim — you have no memory of your own past tool calls otherwise, only your own prior reply text, so reconstructing from memory instead of calling this tool risks getting it wrong.",
|
|
95
|
+
].join("\n"),
|
|
96
|
+
);
|
|
97
|
+
lines.push(
|
|
98
|
+
[
|
|
99
|
+
"You have access to the recall_verbatim tool.",
|
|
100
|
+
"DO:",
|
|
101
|
+
"- If asked about something said in an EARLIER conversation — beyond what you can see in this one — call recall_verbatim to retrieve the actual past messages and quote them, don't reconstruct from memory. This searches a durable archive of what you and this person really said before.",
|
|
102
|
+
"- Use recall_tool_calls, not this, for what you ran in the CURRENT conversation; use recall_verbatim for what was said in past ones.",
|
|
103
|
+
].join("\n"),
|
|
104
|
+
);
|
|
105
|
+
|
|
106
|
+
if (opts.multiUserChannel) {
|
|
107
|
+
// Interim, explicitly non-deterministic mitigation for Mercury replying
|
|
108
|
+
// to every message in a shared space — not a replacement for real
|
|
109
|
+
// @-mention detection, which the registered app's own identity now
|
|
110
|
+
// makes possible but which isn't implemented yet. See NO_REPLY in
|
|
111
|
+
// google-chat-provider.ts for the code side of this check.
|
|
112
|
+
lines.push(
|
|
113
|
+
[
|
|
114
|
+
"This conversation may be a shared space with more than one person, not a private one-on-one chat.",
|
|
115
|
+
"DO:",
|
|
116
|
+
"- Only give a substantive answer if this message is clearly directed at you (e.g. it explicitly mentions/addresses you) or is a direct continuation of an exchange you were already having with this same sender.",
|
|
117
|
+
`- If the message doesn't seem directed at you or isn't relevant to you, respond with exactly \`${NO_REPLY}\` and nothing else — no punctuation, no explanation, nothing before or after it.`,
|
|
118
|
+
].join("\n"),
|
|
119
|
+
);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
lines.push(personaSlot(opts.persona?.tone, DEFAULT_PERSONA_TONE));
|
|
123
|
+
return lines.join("\n");
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Builds the instance's two system prompts from the same plugins and persona:
|
|
128
|
+
* `system` for 1:1 channels (the terminal, the HTTP surface) and `chatSystem`
|
|
129
|
+
* for shared spaces, the only one that carries the multi-user clause — an
|
|
130
|
+
* operator typing normally must never get a NO_REPLY meant for a shared space.
|
|
131
|
+
*/
|
|
132
|
+
export function buildSystemPrompts(opts: {
|
|
133
|
+
pluginFragments: string[];
|
|
134
|
+
skills: Skill[];
|
|
135
|
+
/** Required even when undefined, so a caller can't silently drop the instance's persona. */
|
|
136
|
+
persona: Persona | undefined;
|
|
137
|
+
}): { system: string; chatSystem: string } {
|
|
138
|
+
return {
|
|
139
|
+
system: buildSystemPrompt({ ...opts, multiUserChannel: false }),
|
|
140
|
+
chatSystem: buildSystemPrompt({ ...opts, multiUserChannel: true }),
|
|
141
|
+
};
|
|
142
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Detects, within a single turn's steps, a CLI command call that failed
|
|
3
|
+
* followed later by one for the same binary that succeeded — a candidate
|
|
4
|
+
* procedural correction, distinct from `semantic-fact-extractor.ts` (which
|
|
5
|
+
* is about the user, from `session.messages`) — this is about a *tool*,
|
|
6
|
+
* from the tool-call trace (`StepInfo`), true for whoever uses that
|
|
7
|
+
* command next, not just this user. Pairing is deterministic (pure
|
|
8
|
+
* string/status matching); describing *what* the correction actually was
|
|
9
|
+
* is delegated to the model, same `generateObject`-with-fixed-schema style
|
|
10
|
+
* as `semantic-fact-extractor.ts`.
|
|
11
|
+
*
|
|
12
|
+
* Wired per-turn (`onStepFinish`, see `index.ts`), not from
|
|
13
|
+
* `idle-session-cron.ts`'s idle sweep: the tool-call trace for a turn only
|
|
14
|
+
* exists in memory for the duration of that turn (`tool-log-buffer.ts` is a
|
|
15
|
+
* 200-entry ring buffer shared across every session — not a reliable place
|
|
16
|
+
* to reconstruct one turn's trace minutes or hours later).
|
|
17
|
+
*/
|
|
18
|
+
import { generateObject, type LanguageModel } from "ai";
|
|
19
|
+
import { z } from "zod";
|
|
20
|
+
import type { StepInfo } from "./step-info.ts";
|
|
21
|
+
|
|
22
|
+
export const ProceduralCorrectionCandidateSchema = z.object({ topic: z.string(), value: z.string() });
|
|
23
|
+
export type ProceduralCorrection = { tool: string; topic: string; value: string };
|
|
24
|
+
|
|
25
|
+
type GenerateObjectFn = (params: {
|
|
26
|
+
model: LanguageModel;
|
|
27
|
+
output: "array";
|
|
28
|
+
schema: typeof ProceduralCorrectionCandidateSchema;
|
|
29
|
+
instructions: string;
|
|
30
|
+
prompt: string;
|
|
31
|
+
}) => Promise<{ object: z.infer<typeof ProceduralCorrectionCandidateSchema>[] }>;
|
|
32
|
+
|
|
33
|
+
const SYSTEM_PROMPT =
|
|
34
|
+
"Ti viene mostrato un tentativo di comando fallito e uno, per lo stesso strumento, andato a buon " +
|
|
35
|
+
"fine più tardi nello stesso turno di conversazione. Descrivi la correzione appresa come una coppia " +
|
|
36
|
+
'{topic, value}: "topic" è una chiave breve e stabile per questa specifica correzione (es. ' +
|
|
37
|
+
'"select-prefix", "assignee-operator"), "value" è la regola pratica da ricordare, in una frase, utile ' +
|
|
38
|
+
"per chiunque userà questo comando in futuro — non solo per chi l'ha scoperta ora. Se il secondo " +
|
|
39
|
+
"tentativo non è davvero una correzione dell'errore del primo (es. l'utente ha semplicemente cambiato " +
|
|
40
|
+
"richiesta), restituisci un array vuoto.";
|
|
41
|
+
|
|
42
|
+
/** Same normalization `semantic-fact-extractor.ts` used to apply to identity/preference topics — still needed here since this topic is free text, not a closed enum (procedural corrections are open-ended by nature). */
|
|
43
|
+
function normalizeTopic(topic: string): string {
|
|
44
|
+
return topic
|
|
45
|
+
.trim()
|
|
46
|
+
.toLowerCase()
|
|
47
|
+
.replace(/[\s_]+/g, "-");
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
type Attempt = { command: string; binary: string; ok: boolean; error?: string };
|
|
51
|
+
|
|
52
|
+
function extractAttempts(steps: StepInfo[]): Attempt[] {
|
|
53
|
+
const attempts: Attempt[] = [];
|
|
54
|
+
for (const stepInfo of steps) {
|
|
55
|
+
for (const call of stepInfo.toolCalls) {
|
|
56
|
+
// A CLI attempt is any tool call carrying a `command` string — the
|
|
57
|
+
// generic shape every CLI tool takes, whatever its name (jiraCommand,
|
|
58
|
+
// bitbucketCommand, the residual runCommand). Keying on a fixed tool name
|
|
59
|
+
// would silently miss plugin-owned CLI tools; keying on the input shape
|
|
60
|
+
// stays correct as new CLI plugins are added.
|
|
61
|
+
const input = call.input as { command?: unknown };
|
|
62
|
+
if (typeof input.command !== "string") continue;
|
|
63
|
+
// Group attempts by the binary they invoked, which is just the command's
|
|
64
|
+
// first whitespace-delimited token — no need to fully tokenize the argv
|
|
65
|
+
// (that belongs to whoever executes the command, not to procedural
|
|
66
|
+
// learning). An empty/whitespace-only command has no binary to group by.
|
|
67
|
+
const binary = input.command.trim().split(/\s+/)[0] ?? "";
|
|
68
|
+
if (binary === "") continue;
|
|
69
|
+
|
|
70
|
+
const result = stepInfo.toolResults.find((r) => r.toolCallId === call.toolCallId);
|
|
71
|
+
const output = result?.output as { ok?: unknown; error?: unknown } | undefined;
|
|
72
|
+
if (!output) continue; // no result to learn from (e.g. malformed-args tool-error, not a CLI failure)
|
|
73
|
+
|
|
74
|
+
attempts.push({
|
|
75
|
+
command: input.command,
|
|
76
|
+
binary,
|
|
77
|
+
ok: output.ok === true,
|
|
78
|
+
error: typeof output.error === "string" ? output.error : undefined,
|
|
79
|
+
});
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
return attempts;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** Each failed attempt paired with the first later attempt for the same binary that succeeded — never the reverse, and never across different binaries. */
|
|
86
|
+
function findCorrectionPairs(attempts: Attempt[]): Array<{ failed: Attempt; corrected: Attempt }> {
|
|
87
|
+
const pairs: Array<{ failed: Attempt; corrected: Attempt }> = [];
|
|
88
|
+
for (let i = 0; i < attempts.length; i++) {
|
|
89
|
+
const failed = attempts[i];
|
|
90
|
+
if (!failed || failed.ok) continue;
|
|
91
|
+
const corrected = attempts.slice(i + 1).find((a) => a.binary === failed.binary && a.ok);
|
|
92
|
+
if (corrected) {
|
|
93
|
+
pairs.push({ failed, corrected });
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
return pairs;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Returns a function that extracts candidate procedural corrections from a single turn's steps. */
|
|
100
|
+
export function createToolCorrectionExtractor(
|
|
101
|
+
model: LanguageModel,
|
|
102
|
+
generateObjectFn: GenerateObjectFn = generateObject as unknown as GenerateObjectFn,
|
|
103
|
+
deps?: { log?: (msg: string) => void },
|
|
104
|
+
): (steps: StepInfo[]) => Promise<ProceduralCorrection[]> {
|
|
105
|
+
const log = deps?.log ?? ((msg: string) => console.error(msg));
|
|
106
|
+
|
|
107
|
+
return async (steps) => {
|
|
108
|
+
const pairs = findCorrectionPairs(extractAttempts(steps));
|
|
109
|
+
const corrections: ProceduralCorrection[] = [];
|
|
110
|
+
|
|
111
|
+
for (const { failed, corrected } of pairs) {
|
|
112
|
+
try {
|
|
113
|
+
const { object } = await generateObjectFn({
|
|
114
|
+
model,
|
|
115
|
+
output: "array",
|
|
116
|
+
schema: ProceduralCorrectionCandidateSchema,
|
|
117
|
+
instructions: SYSTEM_PROMPT,
|
|
118
|
+
prompt:
|
|
119
|
+
`Comando fallito: ${failed.command}\n` +
|
|
120
|
+
`Errore: ${failed.error ?? "(nessun messaggio)"}\n` +
|
|
121
|
+
`Comando corretto (riuscito): ${corrected.command}`,
|
|
122
|
+
});
|
|
123
|
+
for (const candidate of object) {
|
|
124
|
+
corrections.push({ tool: failed.binary, topic: normalizeTopic(candidate.topic), value: candidate.value });
|
|
125
|
+
}
|
|
126
|
+
} catch (err) {
|
|
127
|
+
log(`procedural correction extraction failed for ${failed.binary}: ${String(err)}`);
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return corrections;
|
|
132
|
+
};
|
|
133
|
+
}
|