@mercury-fw/core 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +38 -0
- package/dist/index.d.ts +23 -0
- package/dist/src/admin/cli-routes.d.ts +22 -0
- package/dist/src/admin/env-file.d.ts +1 -0
- package/dist/src/admin/model-routes.d.ts +26 -0
- package/dist/src/admin/qdrant-scroll.d.ts +34 -0
- package/dist/src/admin/server.d.ts +40 -0
- package/dist/src/admin/wiki-routes.d.ts +31 -0
- package/dist/src/compose.d.ts +42 -0
- package/dist/src/config/define-config.d.ts +31 -0
- package/dist/src/cron/idle-session-cron.d.ts +80 -0
- package/dist/src/cron/idle-session-scanner.d.ts +16 -0
- package/dist/src/cron/self-review-cron.d.ts +55 -0
- package/dist/src/cron/semantic-consolidation.d.ts +71 -0
- package/dist/src/memory/embedder.d.ts +9 -0
- package/dist/src/memory/episodic-store.d.ts +121 -0
- package/dist/src/memory/memory-provider.d.ts +51 -0
- package/dist/src/memory/semantic-facts-store.d.ts +37 -0
- package/dist/src/memory/tool-corrections-store.d.ts +26 -0
- package/dist/src/memory/verbatim-archive-store.d.ts +86 -0
- package/dist/src/model/client.d.ts +24 -0
- package/dist/src/model/context-size.d.ts +30 -0
- package/dist/src/plugins/manifest.d.ts +29 -0
- package/dist/src/plugins/plugin-loader.d.ts +85 -0
- package/dist/src/router/channel-loader.d.ts +30 -0
- package/dist/src/router/provider.d.ts +7 -0
- package/dist/src/router/terminal-provider.d.ts +37 -0
- package/dist/src/router/terminal.d.ts +41 -0
- package/dist/src/router/tool-log.d.ts +65 -0
- package/dist/src/router/turn-runner.d.ts +86 -0
- package/dist/src/session/agent-turn.d.ts +266 -0
- package/dist/src/session/context-primer.d.ts +16 -0
- package/dist/src/session/episodic-summarizer.d.ts +25 -0
- package/dist/src/session/history.d.ts +95 -0
- package/dist/src/session/pending-confirmation.d.ts +8 -0
- package/dist/src/session/read-skill-tool.d.ts +4 -0
- package/dist/src/session/semantic-fact-extractor.d.ts +45 -0
- package/dist/src/session/step-info.d.ts +24 -0
- package/dist/src/session/summarizer.d.ts +23 -0
- package/dist/src/session/system-prompt.d.ts +38 -0
- package/dist/src/session/tool-correction-extractor.d.ts +43 -0
- package/dist/src/session/tool-log-buffer.d.ts +24 -0
- package/dist/src/session/tool-log-recall-tool.d.ts +18 -0
- package/dist/src/session/tool-start-hook.d.ts +57 -0
- package/dist/src/tools/display-store.d.ts +36 -0
- package/dist/src/tools/present-tool.d.ts +23 -0
- package/dist/src/wiki/frontmatter-schema.d.ts +53 -0
- package/dist/src/wiki/index-entry.d.ts +15 -0
- package/dist/src/wiki/orphan-detector.d.ts +1 -0
- package/dist/src/wiki/self-review-runner.d.ts +48 -0
- package/dist/src/wiki/self-review-tools.d.ts +22 -0
- package/dist/src/wiki/vault-cli.d.ts +2 -0
- package/dist/src/wiki/vault-init.d.ts +7 -0
- package/dist/src/wiki/wiki-note.d.ts +62 -0
- package/dist/src/wiki/wiki-read.d.ts +27 -0
- package/dist/src/wiki/wiki-tools.d.ts +7 -0
- package/index.ts +23 -0
- package/package.json +49 -0
- package/src/admin/cli-routes.ts +48 -0
- package/src/admin/env-file.ts +29 -0
- package/src/admin/model-routes.ts +71 -0
- package/src/admin/public/index.html +416 -0
- package/src/admin/qdrant-scroll.ts +45 -0
- package/src/admin/server.ts +188 -0
- package/src/admin/wiki-routes.ts +93 -0
- package/src/compose.ts +599 -0
- package/src/config/define-config.ts +35 -0
- package/src/cron/.gitkeep +0 -0
- package/src/cron/idle-session-cron.ts +144 -0
- package/src/cron/idle-session-scanner.ts +37 -0
- package/src/cron/self-review-cron.ts +103 -0
- package/src/cron/semantic-consolidation.ts +228 -0
- package/src/memory/.gitkeep +0 -0
- package/src/memory/embedder.ts +15 -0
- package/src/memory/episodic-store.ts +183 -0
- package/src/memory/memory-provider.ts +98 -0
- package/src/memory/semantic-facts-store.ts +89 -0
- package/src/memory/tool-corrections-store.ts +72 -0
- package/src/memory/verbatim-archive-store.ts +202 -0
- package/src/model/client.ts +33 -0
- package/src/model/context-size.ts +42 -0
- package/src/plugins/manifest.ts +47 -0
- package/src/plugins/plugin-loader.ts +205 -0
- package/src/router/channel-loader.ts +56 -0
- package/src/router/provider.ts +7 -0
- package/src/router/terminal-provider.ts +155 -0
- package/src/router/terminal.ts +151 -0
- package/src/router/tool-log.ts +116 -0
- package/src/router/turn-runner.ts +205 -0
- package/src/session/agent-turn.ts +391 -0
- package/src/session/context-primer.ts +134 -0
- package/src/session/episodic-summarizer.ts +38 -0
- package/src/session/history.ts +168 -0
- package/src/session/pending-confirmation.ts +8 -0
- package/src/session/read-skill-tool.ts +38 -0
- package/src/session/semantic-fact-extractor.ts +69 -0
- package/src/session/step-info.ts +27 -0
- package/src/session/summarizer.ts +36 -0
- package/src/session/system-prompt.ts +142 -0
- package/src/session/tool-correction-extractor.ts +133 -0
- package/src/session/tool-log-buffer.ts +73 -0
- package/src/session/tool-log-recall-tool.ts +38 -0
- package/src/session/tool-start-hook.ts +164 -0
- package/src/tools/display-store.ts +89 -0
- package/src/tools/present-tool.ts +41 -0
- package/src/wiki/.gitkeep +0 -0
- package/src/wiki/frontmatter-schema.ts +49 -0
- package/src/wiki/index-entry.ts +59 -0
- package/src/wiki/orphan-detector.ts +61 -0
- package/src/wiki/self-review-runner.ts +133 -0
- package/src/wiki/self-review-tools.ts +162 -0
- package/src/wiki/vault-cli.ts +143 -0
- package/src/wiki/vault-init.ts +43 -0
- package/src/wiki/wiki-note.ts +326 -0
- package/src/wiki/wiki-read.ts +122 -0
- package/src/wiki/wiki-tools.ts +112 -0
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bounded, process-wide history of tool calls across both channels,
|
|
3
|
+
* scoped by session so a recall of "what did you do" only ever surfaces
|
|
4
|
+
* one conversation's own history. Originally built for the admin panel's
|
|
5
|
+
* Tool-log tab (still a consumer, unfiltered); now also backs the
|
|
6
|
+
* `recall_tool_calls` model tool (`tool-log-recall-tool.ts`), which is why
|
|
7
|
+
* this lives in `session/` rather than `admin/` — it's the record of what
|
|
8
|
+
* happened in a session, not an admin-only concern. Nothing else keeps
|
|
9
|
+
* this beyond the current turn (`src/index.ts`'s terminal-only `lastSteps`
|
|
10
|
+
* resets every turn, Google Chat's `onStepFinish` only logs to stderr) —
|
|
11
|
+
* `recordStep` is called additively from both channels' existing
|
|
12
|
+
* `onStepFinish` wiring, changing neither channel's own behavior.
|
|
13
|
+
*/
|
|
14
|
+
import { truncateForDisplay } from "../router/tool-log.ts";
|
|
15
|
+
import type { StepInfo } from "./step-info.ts";
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Whatever the provider that ran the turn calls itself — see
|
|
19
|
+
* `InboundTurn.channel` in `src/router/provider.ts`. Was a closed union
|
|
20
|
+
* ("terminal" | "google-chat") back when the set of providers was fixed in
|
|
21
|
+
* this file; widened so a new provider is a new `Provider` implementation,
|
|
22
|
+
* not an edit here.
|
|
23
|
+
*/
|
|
24
|
+
export type ToolLogChannel = string;
|
|
25
|
+
|
|
26
|
+
export type ToolLogEntry = {
|
|
27
|
+
timestamp: string;
|
|
28
|
+
channel: ToolLogChannel;
|
|
29
|
+
sessionKey: string;
|
|
30
|
+
toolName: string;
|
|
31
|
+
input: string;
|
|
32
|
+
output: string;
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
const MAX_ENTRIES = 200;
|
|
36
|
+
const MAX_CHARS = 2000;
|
|
37
|
+
|
|
38
|
+
let buffer: ToolLogEntry[] = [];
|
|
39
|
+
|
|
40
|
+
export function recordStep(channel: ToolLogChannel, sessionKey: string, step: StepInfo): void {
|
|
41
|
+
for (const call of step.toolCalls) {
|
|
42
|
+
const result = step.toolResults.find((r) => r.toolCallId === call.toolCallId);
|
|
43
|
+
const errorPart = step.content.find((p) => p.type === "tool-error" && p.toolCallId === call.toolCallId);
|
|
44
|
+
const output = result
|
|
45
|
+
? truncateForDisplay(result.output, MAX_CHARS)
|
|
46
|
+
: errorPart
|
|
47
|
+
? `[error] ${truncateForDisplay(errorPart.error, MAX_CHARS)}`
|
|
48
|
+
: "(none)";
|
|
49
|
+
|
|
50
|
+
buffer.push({
|
|
51
|
+
timestamp: new Date().toISOString(),
|
|
52
|
+
channel,
|
|
53
|
+
sessionKey,
|
|
54
|
+
toolName: call.toolName,
|
|
55
|
+
input: truncateForDisplay(call.input, MAX_CHARS),
|
|
56
|
+
output,
|
|
57
|
+
});
|
|
58
|
+
if (buffer.length > MAX_ENTRIES) {
|
|
59
|
+
buffer.shift();
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/** Most recent entries first, optionally restricted to one session. */
|
|
65
|
+
export function getToolLog(filter?: { sessionKey?: string }): ToolLogEntry[] {
|
|
66
|
+
const matching = filter?.sessionKey ? buffer.filter((e) => e.sessionKey === filter.sessionKey) : buffer;
|
|
67
|
+
return [...matching].reverse();
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Test-only: clears the module-level buffer so tests don't leak into each other. */
|
|
71
|
+
export function resetToolLogForTest(): void {
|
|
72
|
+
buffer = [];
|
|
73
|
+
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Model-invocable recall of this conversation's own past tool calls —
|
|
3
|
+
* same split as `wiki-tools.ts` wrapping plain functions in `tool()`.
|
|
4
|
+
* Exists because `SessionHistory` (`history.ts`) only ever persists the
|
|
5
|
+
* final assistant text of a turn, never the tool-call trace: asked to
|
|
6
|
+
* recall exactly what it ran, the model otherwise has no ground truth in
|
|
7
|
+
* its own context and reconstructs a plausible-looking (and sometimes
|
|
8
|
+
* wrong) answer instead of quoting the real one. Backed by
|
|
9
|
+
* `tool-log-buffer.ts`, filtered to the calling session only — never
|
|
10
|
+
* another conversation's history.
|
|
11
|
+
*/
|
|
12
|
+
import type { ExecutableTool } from "@mercury-fw/plugin-types";
|
|
13
|
+
import { tool } from "ai";
|
|
14
|
+
import { z } from "zod";
|
|
15
|
+
import { getToolLog } from "./tool-log-buffer.ts";
|
|
16
|
+
|
|
17
|
+
const DEFAULT_LIMIT = 10;
|
|
18
|
+
const MAX_LIMIT = 50;
|
|
19
|
+
|
|
20
|
+
export type ToolLogRecallDeps = { sessionKey: string };
|
|
21
|
+
|
|
22
|
+
export function createToolLogRecallTool(deps: ToolLogRecallDeps): { recall_tool_calls: ExecutableTool } {
|
|
23
|
+
const recall_tool_calls = tool({
|
|
24
|
+
description:
|
|
25
|
+
"Recall the tool calls (name, input, output) you actually made earlier in THIS conversation, most " +
|
|
26
|
+
"recent first. Use this whenever asked what you ran/queried/did — quote it verbatim, don't reconstruct " +
|
|
27
|
+
"from memory.",
|
|
28
|
+
inputSchema: z.object({
|
|
29
|
+
limit: z.number().int().positive().max(MAX_LIMIT).optional(),
|
|
30
|
+
}),
|
|
31
|
+
execute: async ({ limit }) => {
|
|
32
|
+
const entries = getToolLog({ sessionKey: deps.sessionKey }).slice(0, limit ?? DEFAULT_LIMIT);
|
|
33
|
+
return { ok: true as const, entries };
|
|
34
|
+
},
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
return { recall_tool_calls };
|
|
38
|
+
}
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wraps every tool's `execute` so `onToolStart` fires the moment a call
|
|
3
|
+
* actually begins (before its result is ready), instead of the existing
|
|
4
|
+
* `onStepFinish` visibility (`tool-log.ts`) which only reports after a step
|
|
5
|
+
* already finished, and `onToolFinish` fires once it settles. Used by
|
|
6
|
+
* `src/index.ts`'s `buildTools` for every channel — both terminal and
|
|
7
|
+
* Google Chat always supply a real `onToolStart` (`TurnSink` in
|
|
8
|
+
* `router/provider.ts`), so both are always wrapped.
|
|
9
|
+
*
|
|
10
|
+
* Every wrapped tool's `execute` calls share one `chain` (below), so at
|
|
11
|
+
* most one tool call runs at a time turn-wide — the AI SDK otherwise runs
|
|
12
|
+
* a step's tool calls concurrently (`Promise.all`), which this codebase
|
|
13
|
+
* has never actually relied on; serializing keeps Google Chat's status
|
|
14
|
+
* card for a given `toolCallId` (see `google-chat-provider.ts`) from ever
|
|
15
|
+
* having to handle an overlapping in-flight card.
|
|
16
|
+
*/
|
|
17
|
+
import type { Tool } from "ai";
|
|
18
|
+
|
|
19
|
+
// Elenco piccolo e stabile (6 tool totali oggi) — va aggiornato a mano se
|
|
20
|
+
// si aggiunge un nuovo tool nominato; qualunque nome non elencato qui
|
|
21
|
+
// ricade nel fallback generico, non è un errore. `grep` non è qui: ha una
|
|
22
|
+
// propria etichetta ("Sto cercando…"), non ricade nel bucket "read".
|
|
23
|
+
const WIKI_TOOL_CATEGORY: Record<string, "read" | "write"> = {
|
|
24
|
+
list_files: "read",
|
|
25
|
+
read_file: "read",
|
|
26
|
+
write_file: "write",
|
|
27
|
+
};
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Describes, in one short user-facing sentence, what a tool call is about to do
|
|
31
|
+
* — no arguments/queries shown. A tool a plugin contributed carries its own
|
|
32
|
+
* label via `toolStatusDescribers` (keyed by tool name), so the core never
|
|
33
|
+
* decides how a plugin's tool reads — it just looks the describer up. The other
|
|
34
|
+
* tools are core functionality (wiki, grep, memory) and keep their core-coined
|
|
35
|
+
* labels until they too become internal plugins.
|
|
36
|
+
*/
|
|
37
|
+
export function describeToolStart(
|
|
38
|
+
toolName: string,
|
|
39
|
+
input: unknown,
|
|
40
|
+
toolStatusDescribers: Record<string, (input: unknown) => string>,
|
|
41
|
+
): string {
|
|
42
|
+
const describe = toolStatusDescribers[toolName];
|
|
43
|
+
if (describe) return describe(input);
|
|
44
|
+
if (toolName === "recall_tool_calls") return "Sto consultando la memoria…";
|
|
45
|
+
if (toolName === "read_skill") return "Sto consultando le istruzioni…";
|
|
46
|
+
if (toolName === "grep") return "Sto cercando…";
|
|
47
|
+
const wikiCategory = WIKI_TOOL_CATEGORY[toolName];
|
|
48
|
+
if (wikiCategory === "read") return "Sto leggendo il wiki…";
|
|
49
|
+
if (wikiCategory === "write") return "Sto scrivendo sul wiki…";
|
|
50
|
+
return `Sto usando ${toolName}…`;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
const MAX_DETAIL_CHARS = 300;
|
|
54
|
+
|
|
55
|
+
function truncate(s: string): string {
|
|
56
|
+
return s.length <= MAX_DETAIL_CHARS ? s : `${s.slice(0, MAX_DETAIL_CHARS)}…`;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function stringField(input: unknown, key: string): string | undefined {
|
|
60
|
+
if (typeof input === "object" && input !== null && key in input) {
|
|
61
|
+
const v = (input as Record<string, unknown>)[key];
|
|
62
|
+
if (typeof v === "string") return v;
|
|
63
|
+
}
|
|
64
|
+
return undefined;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* The actual command/input behind a tool call, for display in a status
|
|
69
|
+
* card — `describeToolStart`'s label alone doesn't say *what* ran, only
|
|
70
|
+
* which service. No markdown decoration: Google Chat cards don't render
|
|
71
|
+
* backticks as monospace, so a plain string reads better. Named per-tool
|
|
72
|
+
* (path for a file, pattern for a search, ...) rather than dumping the raw
|
|
73
|
+
* input, since that's what a human actually wants to see; only a genuinely
|
|
74
|
+
* unmapped/future tool falls back to a bounded JSON dump.
|
|
75
|
+
*/
|
|
76
|
+
export function describeToolDetail(toolName: string, input: unknown): string {
|
|
77
|
+
// Any CLI tool carries a `command` string, whatever its name (jiraCommand,
|
|
78
|
+
// bitbucketCommand, the residual runCommand) — show that bare command.
|
|
79
|
+
const command = stringField(input, "command");
|
|
80
|
+
if (command !== undefined) return truncate(command);
|
|
81
|
+
switch (toolName) {
|
|
82
|
+
case "grep":
|
|
83
|
+
return truncate(stringField(input, "pattern") ?? JSON.stringify(input) ?? "");
|
|
84
|
+
case "read_file":
|
|
85
|
+
case "write_file":
|
|
86
|
+
return truncate(stringField(input, "path") ?? JSON.stringify(input) ?? "");
|
|
87
|
+
case "resolve_reference":
|
|
88
|
+
return truncate(stringField(input, "token") ?? JSON.stringify(input) ?? "");
|
|
89
|
+
case "list_files":
|
|
90
|
+
return "Tutti i documenti del wiki";
|
|
91
|
+
case "recall_tool_calls":
|
|
92
|
+
return "Cronologia delle chiamate in questa conversazione";
|
|
93
|
+
default:
|
|
94
|
+
return truncate(JSON.stringify(input) ?? "");
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// `ToolOutcome` is part of the channel `TurnSink` contract, so it lives in
|
|
99
|
+
// `@mercury-fw/channel-types` and is re-exported here for the core callers that
|
|
100
|
+
// import it from this module.
|
|
101
|
+
export type { ToolOutcome } from "@mercury-fw/channel-types";
|
|
102
|
+
import type { ToolOutcome } from "@mercury-fw/channel-types";
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Every tool in this codebase returns a uniform `{ok: true, ...}` /
|
|
106
|
+
* `{ok: false, error, ...}` shape (confirmed across `wiki-tools.ts`,
|
|
107
|
+
* `tool-log-recall-tool.ts`, `cli-tool.ts`), with `cli-tool.ts`'s
|
|
108
|
+
* confirm-required staging additionally setting `pendingConfirmation: true`
|
|
109
|
+
* without having actually run the command yet — so this classifier works
|
|
110
|
+
* for any tool without per-tool special-casing. Defaults to `"success"` for
|
|
111
|
+
* an unrecognized result shape; nothing returned by a tool today hits that
|
|
112
|
+
* branch.
|
|
113
|
+
*/
|
|
114
|
+
export function classifyToolResult(result: unknown): ToolOutcome {
|
|
115
|
+
if (result && typeof result === "object" && "ok" in result) {
|
|
116
|
+
const r = result as { ok?: unknown; pendingConfirmation?: unknown };
|
|
117
|
+
if (r.ok === false) return r.pendingConfirmation === true ? "pending" : "failed";
|
|
118
|
+
if (r.ok === true) return "success";
|
|
119
|
+
}
|
|
120
|
+
return "success";
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Wraps every tool's `execute` to call `onToolStart` first, then run the
|
|
125
|
+
* original, then `onToolFinish` once it settles. `chain` serializes every
|
|
126
|
+
* wrapped tool sharing this one `withToolStartHook` call (i.e. one turn's
|
|
127
|
+
* worth of tools, since `buildTools` builds fresh tools per turn) — see
|
|
128
|
+
* this file's header comment for why.
|
|
129
|
+
*/
|
|
130
|
+
export function withToolStartHook(
|
|
131
|
+
tools: Record<string, Tool>,
|
|
132
|
+
onToolStart: (label: string, detail?: string, toolCallId?: string) => void,
|
|
133
|
+
toolStatusDescribers: Record<string, (input: unknown) => string>,
|
|
134
|
+
onToolFinish?: (toolCallId: string, outcome: ToolOutcome) => void,
|
|
135
|
+
): Record<string, Tool> {
|
|
136
|
+
let chain: Promise<void> = Promise.resolve();
|
|
137
|
+
const wrapped: Record<string, Tool> = {};
|
|
138
|
+
for (const [name, t] of Object.entries(tools)) {
|
|
139
|
+
wrapped[name] = {
|
|
140
|
+
...t,
|
|
141
|
+
execute: (input: unknown, options: { toolCallId: string }) => {
|
|
142
|
+
const label = describeToolStart(name, input, toolStatusDescribers);
|
|
143
|
+
const detail = describeToolDetail(name, input);
|
|
144
|
+
const run = chain.then(async () => {
|
|
145
|
+
onToolStart(label, detail, options.toolCallId);
|
|
146
|
+
try {
|
|
147
|
+
const result = await (t.execute as (i: unknown, o: unknown) => unknown)(input, options);
|
|
148
|
+
onToolFinish?.(options.toolCallId, classifyToolResult(result));
|
|
149
|
+
return result;
|
|
150
|
+
} catch (err) {
|
|
151
|
+
onToolFinish?.(options.toolCallId, "failed");
|
|
152
|
+
throw err;
|
|
153
|
+
}
|
|
154
|
+
});
|
|
155
|
+
chain = run.then(
|
|
156
|
+
() => undefined,
|
|
157
|
+
() => undefined,
|
|
158
|
+
);
|
|
159
|
+
return run;
|
|
160
|
+
},
|
|
161
|
+
} as Tool;
|
|
162
|
+
}
|
|
163
|
+
return wrapped;
|
|
164
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* In-memory staging area for the user-facing `display` artifacts a tool
|
|
3
|
+
* produces (see `ToolDisplay` and the formatter decorator), decoupling
|
|
4
|
+
* *producing* an artifact from *showing* it. A tool `stash`es its already-
|
|
5
|
+
* rendered artifact here and gets back a short `ref`; the model, and only
|
|
6
|
+
* the model, decides whether to actually surface it by calling the
|
|
7
|
+
* `present` tool (see `present-tool.ts`), which marks the ref via
|
|
8
|
+
* `surface`. At finalize the turn runner appends only what
|
|
9
|
+
* `takeSurfaced` returns — the artifacts the model explicitly asked to
|
|
10
|
+
* show — never the whole set unconditionally. The worst case becomes
|
|
11
|
+
* omission (the user says "show me"), never a garbled or force-appended
|
|
12
|
+
* list.
|
|
13
|
+
*
|
|
14
|
+
* Scoped by `sessionKey`, exactly like `confirmation-store.ts`: a ref
|
|
15
|
+
* minted for one session (terminal, or a given Google Chat space+sender)
|
|
16
|
+
* can't be surfaced by another. The store is created once and shared
|
|
17
|
+
* across turns, so a ref stays valid (until its TTL) for a later
|
|
18
|
+
* "show me that list again" follow-up — session-scoped, not turn-scoped.
|
|
19
|
+
*/
|
|
20
|
+
export type DisplayStore = {
|
|
21
|
+
/** Stashes `artifact` for `sessionKey` and returns a fresh ref. */
|
|
22
|
+
stash(sessionKey: string, artifact: string): string;
|
|
23
|
+
/** Marks `ref` to be shown at finalize. Returns `false` if it doesn't
|
|
24
|
+
* exist, belongs to a different session, or has expired. */
|
|
25
|
+
surface(sessionKey: string, ref: string): boolean;
|
|
26
|
+
/** The artifacts surfaced for `sessionKey`, in stash order. Consumes the
|
|
27
|
+
* surfaced flag (so the same artifact isn't re-appended on a later turn
|
|
28
|
+
* unless `surface` is called again) but keeps the entry until its TTL,
|
|
29
|
+
* so a still-valid ref can be surfaced again across turns. */
|
|
30
|
+
takeSurfaced(sessionKey: string): string[];
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
const DEFAULT_TTL_MS = 5 * 60_000;
|
|
34
|
+
|
|
35
|
+
type Entry = { sessionKey: string; artifact: string; surfaced: boolean; expiresAt: number };
|
|
36
|
+
|
|
37
|
+
export function createDisplayStore(
|
|
38
|
+
opts: { now?: () => number; ttlMs?: number; refFn?: () => string } = {},
|
|
39
|
+
): DisplayStore {
|
|
40
|
+
const now = opts.now ?? (() => Date.now());
|
|
41
|
+
const ttlMs = opts.ttlMs ?? DEFAULT_TTL_MS;
|
|
42
|
+
let counter = 0;
|
|
43
|
+
const refFn = opts.refFn ?? (() => `d${++counter}`);
|
|
44
|
+
// Insertion-ordered (Map preserves it) so takeSurfaced returns artifacts
|
|
45
|
+
// in stash order regardless of the order the model surfaced them in.
|
|
46
|
+
const entries = new Map<string, Entry>();
|
|
47
|
+
|
|
48
|
+
return {
|
|
49
|
+
stash(sessionKey, artifact) {
|
|
50
|
+
const ref = refFn();
|
|
51
|
+
entries.set(ref, { sessionKey, artifact, surfaced: false, expiresAt: now() + ttlMs });
|
|
52
|
+
return ref;
|
|
53
|
+
},
|
|
54
|
+
surface(sessionKey, ref) {
|
|
55
|
+
const entry = entries.get(ref);
|
|
56
|
+
if (!entry) {
|
|
57
|
+
return false;
|
|
58
|
+
}
|
|
59
|
+
if (entry.expiresAt <= now()) {
|
|
60
|
+
entries.delete(ref);
|
|
61
|
+
return false;
|
|
62
|
+
}
|
|
63
|
+
if (entry.sessionKey !== sessionKey) {
|
|
64
|
+
return false;
|
|
65
|
+
}
|
|
66
|
+
entry.surfaced = true;
|
|
67
|
+
return true;
|
|
68
|
+
},
|
|
69
|
+
takeSurfaced(sessionKey) {
|
|
70
|
+
const t = now();
|
|
71
|
+
const out: string[] = [];
|
|
72
|
+
// This finalize-time sweep is also where expired entries are reclaimed:
|
|
73
|
+
// the store is a process-lifetime singleton shared across sessions, so
|
|
74
|
+
// without deleting them here an un-surfaced (or never re-surfaced) stash
|
|
75
|
+
// would live forever. Deleting during Map iteration is safe.
|
|
76
|
+
for (const [ref, entry] of entries) {
|
|
77
|
+
if (entry.expiresAt <= t) {
|
|
78
|
+
entries.delete(ref);
|
|
79
|
+
continue;
|
|
80
|
+
}
|
|
81
|
+
if (entry.sessionKey !== sessionKey) continue;
|
|
82
|
+
if (!entry.surfaced) continue;
|
|
83
|
+
out.push(entry.artifact);
|
|
84
|
+
entry.surfaced = false;
|
|
85
|
+
}
|
|
86
|
+
return out;
|
|
87
|
+
},
|
|
88
|
+
};
|
|
89
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `present` tool: the model's explicit "show this artifact to the user"
|
|
3
|
+
* action. A tool that produced a user-facing `display` stashed it in the
|
|
4
|
+
* `DisplayStore` (see `display-store.ts`) and returned a `ref` on the model
|
|
5
|
+
* channel; the model never sees the rendered content, only the ref. Calling
|
|
6
|
+
* `present(ref)` marks that ref to be appended to the reply at finalize.
|
|
7
|
+
*
|
|
8
|
+
* This is what shrinks the model's job from "format this list correctly"
|
|
9
|
+
* (which small models garble) to "show this list? yes/no": the artifact is
|
|
10
|
+
* always the deterministic one the formatter produced, and the model only
|
|
11
|
+
* decides whether it appears. If the model never calls `present`, nothing is
|
|
12
|
+
* shown — the right outcome for a "how many are open?" question answered in
|
|
13
|
+
* prose. Scoped to the calling session, so it can only surface its own refs.
|
|
14
|
+
*/
|
|
15
|
+
import type { ExecutableTool } from "@mercury-fw/plugin-types";
|
|
16
|
+
import { tool } from "ai";
|
|
17
|
+
import { z } from "zod";
|
|
18
|
+
import type { DisplayStore } from "./display-store.ts";
|
|
19
|
+
|
|
20
|
+
export type PresentToolDeps = { sessionKey: string; store: DisplayStore };
|
|
21
|
+
|
|
22
|
+
export function createPresentTool(deps: PresentToolDeps): { present: ExecutableTool } {
|
|
23
|
+
const present = tool({
|
|
24
|
+
description:
|
|
25
|
+
"Show a previously produced artifact (e.g. a list of issues) to the user. Pass the `ref` a prior tool " +
|
|
26
|
+
"result returned (its `displayRef`). Call this ONLY when the user wants to see the artifact itself; if you " +
|
|
27
|
+
"are answering in prose (a count, a yes/no, a single field), do NOT call it — the artifact stays hidden.",
|
|
28
|
+
inputSchema: z.object({ ref: z.string().min(1) }),
|
|
29
|
+
execute: async ({ ref }) => {
|
|
30
|
+
if (!deps.store.surface(deps.sessionKey, ref)) {
|
|
31
|
+
return {
|
|
32
|
+
ok: false as const,
|
|
33
|
+
error: `no artifact with ref "${ref}" to present. Only pass a displayRef returned by a tool result in this conversation.`,
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
return { ok: true as const };
|
|
37
|
+
},
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
return { present };
|
|
41
|
+
}
|
|
File without changes
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Frontmatter schema for wiki notes, based on Open Knowledge
|
|
3
|
+
* Format (OKF — https://cloud.google.com/blog/products/data-analytics/how-the-open-knowledge-format-can-improve-data-sharing):
|
|
4
|
+
* `type` is the only field OKF mandates; `source`/`confidence`/
|
|
5
|
+
* `derived_from`/`last_reviewed` are Mercury-specific extensions, the
|
|
6
|
+
* kind OKF explicitly leaves to the producer. `curated` and `inferred`
|
|
7
|
+
* are validated as separate shapes (discriminated on `type`) because
|
|
8
|
+
* only `inferred` notes carry provenance — a curated doc has no
|
|
9
|
+
* meaningful `confidence` or `derived_from`.
|
|
10
|
+
*/
|
|
11
|
+
import { z } from "zod";
|
|
12
|
+
|
|
13
|
+
export const CuratedFrontmatterSchema = z.object({
|
|
14
|
+
type: z.literal("curated"),
|
|
15
|
+
author: z.string().optional(),
|
|
16
|
+
last_updated: z.string().optional(),
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
export const InferredFrontmatterSchema = z.object({
|
|
20
|
+
type: z.literal("inferred"),
|
|
21
|
+
source: z.literal("agent"),
|
|
22
|
+
confidence: z.enum(["low", "medium", "high"]),
|
|
23
|
+
derived_from: z.array(z.string()).min(1),
|
|
24
|
+
last_reviewed: z.string().nullable(),
|
|
25
|
+
});
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* A fourth category: the lifecycle of one confirm-required action, from
|
|
29
|
+
* staging through its eventual resolution — a deterministic instruction
|
|
30
|
+
* the user explicitly approved via the confirmation-token mechanism,
|
|
31
|
+
* never an autonomous LLM judgment call. Tracks a specific CLI action's
|
|
32
|
+
* own token, written once when staged (`status: "pending"`) and
|
|
33
|
+
* overwritten in place once resolved (`"confirmed"`/`"failed"`). Lives
|
|
34
|
+
* outside `inferred/users/<userId>/` (see `wiki-read.ts`'s
|
|
35
|
+
* `allowedRoots`) so it's structurally invisible to the model's own
|
|
36
|
+
* `list_files`/`grep` — reachable only via the narrow `resolve_reference`
|
|
37
|
+
* tool given the exact token (see `wiki-tools.ts`), never by browsing.
|
|
38
|
+
*/
|
|
39
|
+
export const ConfirmationFrontmatterSchema = z.object({
|
|
40
|
+
type: z.literal("confirmation"),
|
|
41
|
+
status: z.enum(["pending", "confirmed", "failed"]),
|
|
42
|
+
requested_at: z.string(),
|
|
43
|
+
resolved_at: z.string().nullable(),
|
|
44
|
+
command: z.string(),
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
export type CuratedFrontmatter = z.infer<typeof CuratedFrontmatterSchema>;
|
|
48
|
+
export type InferredFrontmatter = z.infer<typeof InferredFrontmatterSchema>;
|
|
49
|
+
export type ConfirmationFrontmatter = z.infer<typeof ConfirmationFrontmatterSchema>;
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic index.md line management. Turns "this curated doc, with
|
|
3
|
+
* this description" into the exact `[[wikilink]]` line format
|
|
4
|
+
* `orphan-detector.ts`'s wikilink check recognizes — instead of trusting
|
|
5
|
+
* the model to freely author (and correctly reproduce, on every edit) the
|
|
6
|
+
* whole file's syntax by hand, which is what `write_index` used to do and
|
|
7
|
+
* how a doc could end up with an index.md line that still isn't
|
|
8
|
+
* recognized as a reference to it.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** Accepts any of "curated/x/y.md", "x/y.md", or "x/y" and normalizes to "x/y" — the form `[[wikilink]]`s use. */
|
|
12
|
+
export function normalizeIndexKey(path: string): string {
|
|
13
|
+
let key = path.trim();
|
|
14
|
+
if (key.startsWith("curated/")) {
|
|
15
|
+
key = key.slice("curated/".length);
|
|
16
|
+
}
|
|
17
|
+
if (key.endsWith(".md")) {
|
|
18
|
+
key = key.slice(0, -".md".length);
|
|
19
|
+
}
|
|
20
|
+
return key;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function escapeRegExp(s: string): string {
|
|
24
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function findEntryLine(lines: string[], key: string): number {
|
|
28
|
+
const re = new RegExp(`\\[\\[${escapeRegExp(key)}(\\||\\])`);
|
|
29
|
+
return lines.findIndex((line) => re.test(line));
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function formatEntry(key: string, description: string): string {
|
|
33
|
+
return `- [[${key}]] — ${description}`;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Adds a line for `key`, or replaces its existing one in place — never duplicates an entry. */
|
|
37
|
+
export function upsertIndexEntry(content: string, key: string, description: string): string {
|
|
38
|
+
const lines = content.length > 0 ? content.split("\n").filter((_, i, arr) => !(i === arr.length - 1 && arr[i] === "")) : [];
|
|
39
|
+
const newLine = formatEntry(key, description);
|
|
40
|
+
|
|
41
|
+
const idx = findEntryLine(lines, key);
|
|
42
|
+
if (idx >= 0) {
|
|
43
|
+
lines[idx] = newLine;
|
|
44
|
+
} else {
|
|
45
|
+
lines.push(newLine);
|
|
46
|
+
}
|
|
47
|
+
return `${lines.join("\n")}\n`;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Removes `key`'s line if present; no-op otherwise. */
|
|
51
|
+
export function removeIndexEntry(content: string, key: string): string {
|
|
52
|
+
const lines = content.length > 0 ? content.split("\n").filter((_, i, arr) => !(i === arr.length - 1 && arr[i] === "")) : [];
|
|
53
|
+
const idx = findEntryLine(lines, key);
|
|
54
|
+
if (idx < 0) {
|
|
55
|
+
return content;
|
|
56
|
+
}
|
|
57
|
+
lines.splice(idx, 1);
|
|
58
|
+
return lines.length > 0 ? `${lines.join("\n")}\n` : "";
|
|
59
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic pre-check for the self-review job (see
|
|
3
|
+
* `self-review-runner.ts`'s index/orphan pass): which curated docs are
|
|
4
|
+
* referenced by neither `index.md` nor a `[[wikilink]]` from another
|
|
5
|
+
* curated doc. Detection only — what to do about an orphan (add it to
|
|
6
|
+
* the index, add a cross-link, both) is left to the LLM's judgment;
|
|
7
|
+
* that decision can't be made deterministically.
|
|
8
|
+
*/
|
|
9
|
+
import { basename, resolve } from "node:path";
|
|
10
|
+
import { listWikiFilesInRoots, readWikiFileInRoots, readIndexFile } from "./wiki-read.ts";
|
|
11
|
+
|
|
12
|
+
/** Both `[[jira-fields]]` and `[[jira-fields|display text]]` resolve to "jira-fields". */
|
|
13
|
+
function extractWikilinks(content: string): Set<string> {
|
|
14
|
+
const links = new Set<string>();
|
|
15
|
+
const regex = /\[\[([^\]|]+)(?:\|[^\]]*)?\]\]/g;
|
|
16
|
+
let match: RegExpExecArray | null;
|
|
17
|
+
while ((match = regex.exec(content))) {
|
|
18
|
+
links.add(match[1]!.trim());
|
|
19
|
+
}
|
|
20
|
+
return links;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export async function findOrphanCuratedDocs(vaultPath: string): Promise<string[]> {
|
|
24
|
+
const curatedRoot = resolve(vaultPath, "curated");
|
|
25
|
+
const curatedFiles = await listWikiFilesInRoots(vaultPath, [curatedRoot]);
|
|
26
|
+
const indexContent = await readIndexFile(vaultPath);
|
|
27
|
+
|
|
28
|
+
const fileContents = new Map<string, string>();
|
|
29
|
+
for (const file of curatedFiles) {
|
|
30
|
+
fileContents.set(file, await readWikiFileInRoots(vaultPath, [curatedRoot], file));
|
|
31
|
+
}
|
|
32
|
+
const indexLinks = extractWikilinks(indexContent);
|
|
33
|
+
|
|
34
|
+
const orphans: string[] = [];
|
|
35
|
+
for (const file of curatedFiles) {
|
|
36
|
+
const curatedRelative = file.slice("curated/".length).replace(/\.md$/, "");
|
|
37
|
+
const base = basename(curatedRelative);
|
|
38
|
+
// index.md can mention a doc either as a plain path or as a [[wikilink]]
|
|
39
|
+
// (the Karpathy-pattern index is itself just markdown with wikilinks).
|
|
40
|
+
const referencedByIndex =
|
|
41
|
+
indexContent.includes(file) || indexLinks.has(curatedRelative) || indexLinks.has(base);
|
|
42
|
+
|
|
43
|
+
// "Referenced" means linked from ANOTHER doc — a doc's own [[self-link]]
|
|
44
|
+
// doesn't count, or every self-linking doc would wrongly stop being orphaned.
|
|
45
|
+
let referencedByLink = false;
|
|
46
|
+
for (const [otherFile, content] of fileContents) {
|
|
47
|
+
if (otherFile === file) continue;
|
|
48
|
+
const links = extractWikilinks(content);
|
|
49
|
+
if (links.has(curatedRelative) || links.has(base)) {
|
|
50
|
+
referencedByLink = true;
|
|
51
|
+
break;
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
if (!referencedByIndex && !referencedByLink) {
|
|
56
|
+
orphans.push(file);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
return orphans.sort();
|
|
61
|
+
}
|