peon-mem 1.0.7 → 1.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/peon-mem.mjs +9 -2
- package/dist/compression.js +4 -3
- package/dist/config.d.ts +13 -0
- package/dist/config.js +26 -0
- package/dist/daemon.d.ts +0 -3
- package/dist/daemon.js +12 -137
- package/dist/embedding-store.d.ts +10 -0
- package/dist/embedding-store.js +97 -19
- package/dist/entity-extraction.js +4 -3
- package/dist/global-extraction.js +4 -3
- package/dist/hyde.js +4 -3
- package/dist/logger.d.ts +20 -0
- package/dist/logger.js +73 -6
- package/dist/memory-store.d.ts +5 -1
- package/dist/memory-store.js +158 -12
- package/dist/monitor.js +14 -224
- package/dist/overview.d.ts +18 -6
- package/dist/overview.js +92 -14
- package/dist/processor.d.ts +36 -0
- package/dist/processor.js +85 -2
- package/dist/quality.d.ts +17 -1
- package/dist/quality.js +152 -25
- package/dist/recuration.js +4 -3
- package/dist/reranker.js +4 -3
- package/dist/tools.d.ts +13 -0
- package/dist/tools.js +76 -14
- package/package.json +3 -3
- package/scripts/peon-operations-watch.mjs +80 -0
- package/dist/roundtable.d.ts +0 -104
- package/dist/roundtable.js +0 -787
package/bin/peon-mem.mjs
CHANGED
|
@@ -25,12 +25,16 @@ const DAEMON = join(PKG, "dist", "daemon-cli.js");
|
|
|
25
25
|
const MCP = join(PKG, "dist", "index.js");
|
|
26
26
|
const NODE = process.execPath;
|
|
27
27
|
|
|
28
|
-
|
|
28
|
+
// An MCP client launches `npx -y peon-mem` with no arguments and stdin as a pipe; that is
|
|
29
|
+
// exactly what the MCP registry entry and marketplaces install. Serve MCP then. A person at a
|
|
30
|
+
// terminal (stdin is a TTY) still gets the help text.
|
|
31
|
+
const cmd = process.argv[2] || (process.stdin.isTTY ? "help" : "mcp");
|
|
29
32
|
const DRY = process.argv.includes("--dry-run");
|
|
30
33
|
const YES = process.argv.includes("--yes") || !process.stdout.isTTY;
|
|
31
34
|
const log = (s) => console.log(s);
|
|
32
35
|
const act = (desc, fn) => { log((DRY ? " [dry-run] " : " ✔ ") + desc); if (!DRY) fn(); };
|
|
33
|
-
|
|
36
|
+
// In MCP mode stdin/stdout carry JSON-RPC, so nothing else may read stdin or write stdout.
|
|
37
|
+
const rl = YES || cmd === "mcp" ? null : createInterface({ input: process.stdin, output: process.stdout });
|
|
34
38
|
async function ask(q, def) {
|
|
35
39
|
if (!rl) return def;
|
|
36
40
|
const a = (await rl.question(`${q}${def ? ` [${def}]` : ""}: `)).trim();
|
|
@@ -306,6 +310,8 @@ if (cmd === "install") {
|
|
|
306
310
|
log("Remove [mcp_servers.peon] / mcpServers.peon from Codex/Gemini/Cursor/Cline configs if you added them.");
|
|
307
311
|
log("Memory data untouched: <project>/.peon/ and " + DEFAULT_HOME);
|
|
308
312
|
rl?.close();
|
|
313
|
+
} else if (cmd === "mcp") {
|
|
314
|
+
await import(MCP);
|
|
309
315
|
} else if (cmd === "daemon") {
|
|
310
316
|
await import(DAEMON);
|
|
311
317
|
} else if (cmd === "doctor") {
|
|
@@ -319,6 +325,7 @@ if (cmd === "install") {
|
|
|
319
325
|
log("peon-mem — memory brain for AI coding agents");
|
|
320
326
|
log(" peon-mem install [--yes] [--dry-run] guided setup (memory home → LLM → daemon → apps)");
|
|
321
327
|
log(" peon-mem uninstall remove service + hooks (data stays)");
|
|
328
|
+
log(" peon-mem mcp run the MCP server over stdio (what MCP clients launch)");
|
|
322
329
|
log(" peon-mem daemon run daemon in foreground");
|
|
323
330
|
log(" peon-mem doctor health check");
|
|
324
331
|
rl?.close();
|
package/dist/compression.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
|
|
1
2
|
/**
|
|
2
3
|
* Builds the LLM summarizer the brain uses to compress a topic cluster into one
|
|
3
4
|
* gist belief. Kept separate from brain.ts so the curation logic stays pure and
|
|
@@ -5,7 +6,7 @@
|
|
|
5
6
|
* or no key is configured (the brain then runs cost-free, skipping compression).
|
|
6
7
|
*/
|
|
7
8
|
export function createClusterSummarizer(config) {
|
|
8
|
-
if (
|
|
9
|
+
if (!llmEnabled(config))
|
|
9
10
|
return null;
|
|
10
11
|
return async (cluster) => {
|
|
11
12
|
const beliefs = cluster.members.map((m, i) => `${i + 1}. ${m.content}`).join("\n");
|
|
@@ -14,9 +15,9 @@ export function createClusterSummarizer(config) {
|
|
|
14
15
|
"Losing a specific fact is a failure; merge wording, never drop information. Drop only redundancy and filler. " +
|
|
15
16
|
"Output ONLY the summary sentence(s) — no preamble, no markdown, no quotes around it. Max 240 characters.";
|
|
16
17
|
const user = `Topic: ${cluster.entity}\n\nBeliefs to compress:\n${beliefs}\n\nOne compact summary:`;
|
|
17
|
-
const response = await fetch(
|
|
18
|
+
const response = await fetch(llmEndpoint(config), {
|
|
18
19
|
method: "POST",
|
|
19
|
-
headers:
|
|
20
|
+
headers: llmHeaders(config),
|
|
20
21
|
body: JSON.stringify({
|
|
21
22
|
model: config.processingModel,
|
|
22
23
|
messages: [
|
package/dist/config.d.ts
CHANGED
|
@@ -19,4 +19,17 @@ export interface PeonConfig {
|
|
|
19
19
|
type Env = Record<string, string | undefined>;
|
|
20
20
|
export declare function loadPeonConfig(env?: Env): PeonConfig;
|
|
21
21
|
export declare function readEnvFile(startDir?: string): Env;
|
|
22
|
+
/**
|
|
23
|
+
* Is an LLM available for optional AI passes (compression, entity extraction, HyDE,
|
|
24
|
+
* global extraction)?
|
|
25
|
+
*
|
|
26
|
+
* These used to gate on `openRouterApiKey`, which meant a fully-local setup
|
|
27
|
+
* (PEON_PROVIDER=ollama, no OpenRouter key) silently skipped every one of them —
|
|
28
|
+
* "local mode" was not actually local. A local provider needs no key; a hosted one does.
|
|
29
|
+
*/
|
|
30
|
+
export declare function llmEnabled(config: PeonConfig): boolean;
|
|
31
|
+
/** The chat-completions endpoint for the configured provider. */
|
|
32
|
+
export declare function llmEndpoint(config: PeonConfig): string;
|
|
33
|
+
/** Auth + content headers for the configured provider (local providers need no key). */
|
|
34
|
+
export declare function llmHeaders(config: PeonConfig): Record<string, string>;
|
|
22
35
|
export {};
|
package/dist/config.js
CHANGED
|
@@ -97,3 +97,29 @@ function findEnvFile(startDir) {
|
|
|
97
97
|
current = dirname(current);
|
|
98
98
|
}
|
|
99
99
|
}
|
|
100
|
+
/**
|
|
101
|
+
* Is an LLM available for optional AI passes (compression, entity extraction, HyDE,
|
|
102
|
+
* global extraction)?
|
|
103
|
+
*
|
|
104
|
+
* These used to gate on `openRouterApiKey`, which meant a fully-local setup
|
|
105
|
+
* (PEON_PROVIDER=ollama, no OpenRouter key) silently skipped every one of them —
|
|
106
|
+
* "local mode" was not actually local. A local provider needs no key; a hosted one does.
|
|
107
|
+
*/
|
|
108
|
+
export function llmEnabled(config) {
|
|
109
|
+
if (config.aiMode === "off")
|
|
110
|
+
return false;
|
|
111
|
+
if (config.provider === "ollama")
|
|
112
|
+
return true;
|
|
113
|
+
return Boolean(config.llmApiKey ?? config.openRouterApiKey);
|
|
114
|
+
}
|
|
115
|
+
/** The chat-completions endpoint for the configured provider. */
|
|
116
|
+
export function llmEndpoint(config) {
|
|
117
|
+
return `${config.llmBaseUrl.replace(/\/$/, "")}/chat/completions`;
|
|
118
|
+
}
|
|
119
|
+
/** Auth + content headers for the configured provider (local providers need no key). */
|
|
120
|
+
export function llmHeaders(config) {
|
|
121
|
+
return {
|
|
122
|
+
Authorization: `Bearer ${config.llmApiKey ?? config.openRouterApiKey ?? ""}`,
|
|
123
|
+
"Content-Type": "application/json"
|
|
124
|
+
};
|
|
125
|
+
}
|
package/dist/daemon.d.ts
CHANGED
|
@@ -1,11 +1,8 @@
|
|
|
1
|
-
import { type RoundtableAgentRunner } from "./roundtable.js";
|
|
2
1
|
export interface StartPeonDaemonOptions {
|
|
3
2
|
host?: string;
|
|
4
3
|
port?: number;
|
|
5
4
|
logDir?: string;
|
|
6
5
|
globalMemoryDir?: string;
|
|
7
|
-
/** Override the local Codex/Claude bridge. Intended for tests and custom launchers. */
|
|
8
|
-
roundtableRunner?: RoundtableAgentRunner;
|
|
9
6
|
}
|
|
10
7
|
export interface PeonDaemonHandle {
|
|
11
8
|
host: string;
|
package/dist/daemon.js
CHANGED
|
@@ -7,7 +7,6 @@ import { URL } from "node:url";
|
|
|
7
7
|
import { PeonLogger } from "./logger.js";
|
|
8
8
|
import { renderMonitorHtml } from "./monitor.js";
|
|
9
9
|
import { renderTokenAbMonitorHtml } from "./token-ab-monitor.js";
|
|
10
|
-
import { RoundtableManager } from "./roundtable.js";
|
|
11
10
|
import { SessionIndex } from "./session-index.js";
|
|
12
11
|
import { createPeonTools } from "./tools.js";
|
|
13
12
|
import { summarizeBeliefs, detectDuplicates, computeTokenSavings, enrichInjection, filterStrayProjects } from "./overview.js";
|
|
@@ -103,9 +102,9 @@ export function canonicalProjectPath(projectPath, home = homedir()) {
|
|
|
103
102
|
/**
|
|
104
103
|
* Reject requests that aren't from a loopback caller — defeats DNS-rebinding and drive-by-localhost
|
|
105
104
|
* attacks where a malicious web page POSTs to the daemon (which would otherwise write/read a brain
|
|
106
|
-
* at an attacker-controlled path). A bad Host header (rebinding)
|
|
107
|
-
* (browser drive-by) is refused. The node hook (no
|
|
108
|
-
* Origin)
|
|
105
|
+
* at an attacker-controlled path). A bad Host header (rebinding), a cross-origin Origin/Referer
|
|
106
|
+
* (browser drive-by) or a browser's Sec-Fetch-Site: cross-site is refused. The node hook (no
|
|
107
|
+
* Origin) and the local monitor UI (loopback Origin, same-origin fetches) all pass.
|
|
109
108
|
*/
|
|
110
109
|
function isLoopbackHostname(hostname) {
|
|
111
110
|
const h = hostname.toLowerCase().replace(/^\[|\]$/g, "");
|
|
@@ -129,6 +128,12 @@ function isLocalRequest(request) {
|
|
|
129
128
|
// A present Host must be loopback (a missing Host — rare, HTTP/1.0 — is allowed; bind is 127.0.0.1).
|
|
130
129
|
if (rawHost !== undefined && !isLoopbackHostname(hostnameFromHostHeader(String(rawHost))))
|
|
131
130
|
return false;
|
|
131
|
+
// Browsers label every request with where it came from. An <img> or no-referrer fetch from
|
|
132
|
+
// another site carries no Origin or Referer, so the loop below cannot see it, yet a GET like
|
|
133
|
+
// /context?projectPath=... still creates a brain at that path. Non-browser callers (the hook,
|
|
134
|
+
// the MCP server, curl) send no Sec-Fetch-Site and are unaffected.
|
|
135
|
+
if (String(request.headers["sec-fetch-site"] ?? "").toLowerCase() === "cross-site")
|
|
136
|
+
return false;
|
|
132
137
|
for (const header of [request.headers.origin, request.headers.referer]) {
|
|
133
138
|
if (!header)
|
|
134
139
|
continue;
|
|
@@ -159,53 +164,6 @@ export async function startPeonDaemon(options = {}) {
|
|
|
159
164
|
const tools = createPeonTools({ globalMemoryDir: options.globalMemoryDir, sessionIndexPath });
|
|
160
165
|
const logger = new PeonLogger({ logDir: options.logDir });
|
|
161
166
|
const projectRegistry = new ProjectRegistry(stateDir);
|
|
162
|
-
const roundtable = new RoundtableManager({
|
|
163
|
-
stateDir: join(stateDir, "roundtables"),
|
|
164
|
-
runner: options.roundtableRunner,
|
|
165
|
-
memory: {
|
|
166
|
-
getContext: (input) => tools.getContext(input),
|
|
167
|
-
saveProposal: async (projectPath, question, proposal) => {
|
|
168
|
-
const session = await tools.startSession({ projectPath, client: "peon-roundtable", cwd: projectPath });
|
|
169
|
-
try {
|
|
170
|
-
await tools.recordEvent({
|
|
171
|
-
sessionId: session.sessionId,
|
|
172
|
-
type: "roundtable_proposal",
|
|
173
|
-
content: `Roundtable proposal for “${question.slice(0, 220)}”: ${proposal.slice(0, 6_000)}`
|
|
174
|
-
});
|
|
175
|
-
}
|
|
176
|
-
finally {
|
|
177
|
-
await tools.endSession({ sessionId: session.sessionId }).catch(() => undefined);
|
|
178
|
-
}
|
|
179
|
-
},
|
|
180
|
-
saveDecision: async (projectPath, question, decision) => {
|
|
181
|
-
const session = await tools.startSession({ projectPath, client: "peon-roundtable", cwd: projectPath });
|
|
182
|
-
try {
|
|
183
|
-
await tools.recordEvent({
|
|
184
|
-
sessionId: session.sessionId,
|
|
185
|
-
type: "decision",
|
|
186
|
-
content: `Roundtable decision for “${question.slice(0, 220)}”: ${decision.slice(0, 6_000)}`
|
|
187
|
-
});
|
|
188
|
-
}
|
|
189
|
-
finally {
|
|
190
|
-
await tools.endSession({ sessionId: session.sessionId }).catch(() => undefined);
|
|
191
|
-
}
|
|
192
|
-
},
|
|
193
|
-
saveResult: async (projectPath, question, result) => {
|
|
194
|
-
const session = await tools.startSession({ projectPath, client: "peon-roundtable", cwd: projectPath });
|
|
195
|
-
try {
|
|
196
|
-
await tools.recordEvent({
|
|
197
|
-
sessionId: session.sessionId,
|
|
198
|
-
type: "roundtable_result",
|
|
199
|
-
content: `Roundtable implementation result for “${question.slice(0, 220)}”: ${result.slice(0, 6_000)}`
|
|
200
|
-
});
|
|
201
|
-
}
|
|
202
|
-
finally {
|
|
203
|
-
await tools.endSession({ sessionId: session.sessionId }).catch(() => undefined);
|
|
204
|
-
}
|
|
205
|
-
}
|
|
206
|
-
}
|
|
207
|
-
});
|
|
208
|
-
await roundtable.initialize();
|
|
209
167
|
const sessionIndex = new SessionIndex(sessionIndexPath);
|
|
210
168
|
const activeSessions = new Map();
|
|
211
169
|
// Clear zombie sessions left behind by a crashed run before rehydrating, so the
|
|
@@ -251,8 +209,7 @@ export async function startPeonDaemon(options = {}) {
|
|
|
251
209
|
sendJson(response, 403, { error: "forbidden: non-local request origin" });
|
|
252
210
|
return;
|
|
253
211
|
}
|
|
254
|
-
const noisy = NOISY_LOG_PATHS.has(requestUrl.pathname)
|
|
255
|
-
(requestMethod === "GET" && requestUrl.pathname.startsWith("/roundtable/"));
|
|
212
|
+
const noisy = NOISY_LOG_PATHS.has(requestUrl.pathname);
|
|
256
213
|
if (!noisy) {
|
|
257
214
|
await logger.log("request_in", {
|
|
258
215
|
requestId,
|
|
@@ -262,36 +219,7 @@ export async function startPeonDaemon(options = {}) {
|
|
|
262
219
|
});
|
|
263
220
|
}
|
|
264
221
|
try {
|
|
265
|
-
|
|
266
|
-
const rawProjectPath = requestUrl.searchParams.get("projectPath");
|
|
267
|
-
if (!rawProjectPath)
|
|
268
|
-
throw new BadRequestError("projectPath is required");
|
|
269
|
-
const projectPath = canonicalProjectPath(rawProjectPath);
|
|
270
|
-
response.writeHead(200, {
|
|
271
|
-
"content-type": "text/event-stream; charset=utf-8",
|
|
272
|
-
"cache-control": "no-cache, no-transform",
|
|
273
|
-
connection: "keep-alive",
|
|
274
|
-
"x-accel-buffering": "no"
|
|
275
|
-
});
|
|
276
|
-
response.write(": thesis room connected\n\n");
|
|
277
|
-
const unsubscribe = roundtable.subscribe((run) => {
|
|
278
|
-
if (run.projectPath !== projectPath || response.writableEnded)
|
|
279
|
-
return;
|
|
280
|
-
response.write(`data: ${JSON.stringify(run)}\n\n`);
|
|
281
|
-
});
|
|
282
|
-
const heartbeat = setInterval(() => {
|
|
283
|
-
if (!response.writableEnded)
|
|
284
|
-
response.write(": keep-alive\n\n");
|
|
285
|
-
}, 20_000);
|
|
286
|
-
if (typeof heartbeat.unref === "function")
|
|
287
|
-
heartbeat.unref();
|
|
288
|
-
request.on("close", () => {
|
|
289
|
-
clearInterval(heartbeat);
|
|
290
|
-
unsubscribe();
|
|
291
|
-
});
|
|
292
|
-
return;
|
|
293
|
-
}
|
|
294
|
-
const result = await routeRequest(request, tools, activeSessions, monitorState, logger, projectRegistry, roundtable);
|
|
222
|
+
const result = await routeRequest(request, tools, activeSessions, monitorState, logger, projectRegistry);
|
|
295
223
|
if ("html" in result) {
|
|
296
224
|
sendHtml(response, result.status ?? 200, result.html);
|
|
297
225
|
}
|
|
@@ -422,15 +350,12 @@ async function maybeRecurate(tools, monitorState, logger) {
|
|
|
422
350
|
await logger.log("recurate_fail", { projectPath, error: error instanceof Error ? error.message : "unknown" });
|
|
423
351
|
}
|
|
424
352
|
}
|
|
425
|
-
async function routeRequest(request, tools, activeSessions, monitorState, logger, projectRegistry
|
|
353
|
+
async function routeRequest(request, tools, activeSessions, monitorState, logger, projectRegistry) {
|
|
426
354
|
const url = new URL(request.url ?? "/", "http://127.0.0.1");
|
|
427
355
|
const method = request.method ?? "GET";
|
|
428
356
|
if (method === "GET" && url.pathname === "/health") {
|
|
429
357
|
return { body: { ok: true, service: "peon-daemon" } };
|
|
430
358
|
}
|
|
431
|
-
if (method === "GET" && url.pathname === "/favicon.ico") {
|
|
432
|
-
return { status: 204, body: "" };
|
|
433
|
-
}
|
|
434
359
|
if ((method === "GET" || method === "HEAD") && url.pathname === "/monitor") {
|
|
435
360
|
return { html: renderMonitorHtml() };
|
|
436
361
|
}
|
|
@@ -440,56 +365,6 @@ async function routeRequest(request, tools, activeSessions, monitorState, logger
|
|
|
440
365
|
if (method === "GET" && url.pathname === "/monitor/state") {
|
|
441
366
|
return { body: await buildMonitorState(tools, activeSessions, monitorState, logger) };
|
|
442
367
|
}
|
|
443
|
-
if (method === "GET" && url.pathname === "/roundtable/status") {
|
|
444
|
-
return { body: { agents: await roundtable.availability() } };
|
|
445
|
-
}
|
|
446
|
-
if (method === "GET" && url.pathname === "/roundtable/runs") {
|
|
447
|
-
const rawProjectPath = url.searchParams.get("projectPath");
|
|
448
|
-
const projectPath = rawProjectPath ? canonicalProjectPath(rawProjectPath) : undefined;
|
|
449
|
-
return { body: { runs: roundtable.list(projectPath).slice(0, 30) } };
|
|
450
|
-
}
|
|
451
|
-
const roundtableRunMatch = url.pathname.match(/^\/roundtable\/runs\/([^/]+)$/);
|
|
452
|
-
if (method === "GET" && roundtableRunMatch) {
|
|
453
|
-
const run = roundtable.get(decodeURIComponent(roundtableRunMatch[1]));
|
|
454
|
-
if (!run)
|
|
455
|
-
return { status: 404, body: { error: "Roundtable discussion not found" } };
|
|
456
|
-
return { body: run };
|
|
457
|
-
}
|
|
458
|
-
if (method === "POST" && url.pathname === "/roundtable/runs") {
|
|
459
|
-
const input = await readJson(request);
|
|
460
|
-
if (!input.projectPath)
|
|
461
|
-
throw new BadRequestError("projectPath is required");
|
|
462
|
-
const projectPath = canonicalProjectPath(input.projectPath);
|
|
463
|
-
if (!monitorState.knownProjects.has(projectPath) && !isRealProjectPath(projectPath)) {
|
|
464
|
-
throw new BadRequestError("Select a project already known to Peon");
|
|
465
|
-
}
|
|
466
|
-
const availability = await roundtable.availability();
|
|
467
|
-
const missing = ["codex", "claude"].filter((agent) => !availability[agent].available);
|
|
468
|
-
if (missing.length)
|
|
469
|
-
throw new BadRequestError(`Local client unavailable: ${missing.join(" and ")}`);
|
|
470
|
-
await rememberProject(monitorState, projectRegistry, projectPath);
|
|
471
|
-
return { status: 202, body: await roundtable.start(projectPath, input.question) };
|
|
472
|
-
}
|
|
473
|
-
const roundtableApprovalMatch = url.pathname.match(/^\/roundtable\/runs\/([^/]+)\/(approve|reject)$/);
|
|
474
|
-
if (method === "POST" && roundtableApprovalMatch) {
|
|
475
|
-
const id = decodeURIComponent(roundtableApprovalMatch[1]);
|
|
476
|
-
const action = roundtableApprovalMatch[2];
|
|
477
|
-
return { body: action === "approve" ? await roundtable.approve(id) : await roundtable.reject(id) };
|
|
478
|
-
}
|
|
479
|
-
const roundtableProposalMatch = url.pathname.match(/^\/roundtable\/runs\/([^/]+)\/propose$/);
|
|
480
|
-
if (method === "POST" && roundtableProposalMatch) {
|
|
481
|
-
return { body: await roundtable.propose(decodeURIComponent(roundtableProposalMatch[1])) };
|
|
482
|
-
}
|
|
483
|
-
const roundtableContinueMatch = url.pathname.match(/^\/roundtable\/runs\/([^/]+)\/continue$/);
|
|
484
|
-
if (method === "POST" && roundtableContinueMatch) {
|
|
485
|
-
return { body: await roundtable.continueDiscussion(decodeURIComponent(roundtableContinueMatch[1])) };
|
|
486
|
-
}
|
|
487
|
-
const roundtableMessageMatch = url.pathname.match(/^\/roundtable\/runs\/([^/]+)\/message$/);
|
|
488
|
-
if (method === "POST" && roundtableMessageMatch) {
|
|
489
|
-
const id = decodeURIComponent(roundtableMessageMatch[1]);
|
|
490
|
-
const input = await readJson(request);
|
|
491
|
-
return { body: await roundtable.message(id, input.message) };
|
|
492
|
-
}
|
|
493
368
|
if (method === "GET" && url.pathname === "/logs") {
|
|
494
369
|
const limit = Number.parseInt(url.searchParams.get("limit") ?? "100", 10);
|
|
495
370
|
return { body: { entries: await logger.recent(Number.isFinite(limit) ? limit : 100) } };
|
|
@@ -20,6 +20,14 @@ export interface SyncResult {
|
|
|
20
20
|
reused: number;
|
|
21
21
|
pruned: number;
|
|
22
22
|
}
|
|
23
|
+
export declare function embeddingCacheStats(): {
|
|
24
|
+
cachedStores: number;
|
|
25
|
+
diskReads: number;
|
|
26
|
+
reads: number;
|
|
27
|
+
expireOlderThan: (now: number) => void;
|
|
28
|
+
};
|
|
29
|
+
/** Test helper: forget every cached sidecar. */
|
|
30
|
+
export declare function resetEmbeddingCaches(): void;
|
|
23
31
|
/** Test helper: forget learned widths, simulating a fresh daemon process. */
|
|
24
32
|
export declare function resetEmbeddingDimensionCache(): void;
|
|
25
33
|
export declare class EmbeddingStore {
|
|
@@ -35,6 +43,8 @@ export declare class EmbeddingStore {
|
|
|
35
43
|
* empty map rather than throwing (retrieval falls back to lexical-only).
|
|
36
44
|
*/
|
|
37
45
|
sync(records: MemoryRecord[], client: EmbeddingClient | null): Promise<SyncResult>;
|
|
46
|
+
/** Release this store's parsed sidecar. Costs one re-read, never correctness. */
|
|
47
|
+
dropCache(): void;
|
|
38
48
|
/** Read vectors without recomputing — used by read-only retrieval paths. */
|
|
39
49
|
vectorById(): Promise<Map<string, EmbeddingVector>>;
|
|
40
50
|
private persist;
|
package/dist/embedding-store.js
CHANGED
|
@@ -1,6 +1,67 @@
|
|
|
1
1
|
import { mkdir, readFile, rename, stat, writeFile } from "node:fs/promises";
|
|
2
2
|
import { dirname, join } from "node:path";
|
|
3
3
|
import { contentHash } from "./embeddings.js";
|
|
4
|
+
/**
|
|
5
|
+
* Bounded registry of live sidecar caches.
|
|
6
|
+
*
|
|
7
|
+
* The cache below keeps a whole embeddings.jsonl parsed in memory so read-only
|
|
8
|
+
* retrieval doesn't re-parse a multi-MB file on every prompt. That is a real win,
|
|
9
|
+
* but the daemon holds one store per project and nothing evicted: a heap snapshot
|
|
10
|
+
* showed 8 stores pinning 60,288 Float32Array vectors / 338 MB of native backing,
|
|
11
|
+
* with RSS past 1.9 GB — enough GC pressure to peg the CPU and stop the daemon
|
|
12
|
+
* answering. So caches are now LRU-bounded and released once idle; a dropped cache
|
|
13
|
+
* costs one re-read, never correctness.
|
|
14
|
+
*/
|
|
15
|
+
const MAX_CACHED_STORES = Number(process.env.PEON_EMBED_CACHE_STORES) > 0
|
|
16
|
+
? Number(process.env.PEON_EMBED_CACHE_STORES)
|
|
17
|
+
: 2;
|
|
18
|
+
const CACHE_TTL_MS = Number(process.env.PEON_EMBED_CACHE_TTL_MS) > 0
|
|
19
|
+
? Number(process.env.PEON_EMBED_CACHE_TTL_MS)
|
|
20
|
+
: 5 * 60 * 1000;
|
|
21
|
+
/** Insertion order is LRU order: least-recently-used first. */
|
|
22
|
+
const liveCaches = new Map();
|
|
23
|
+
let diskReads = 0;
|
|
24
|
+
function touchCache(store, at) {
|
|
25
|
+
liveCaches.delete(store);
|
|
26
|
+
liveCaches.set(store, at);
|
|
27
|
+
pruneCaches(at);
|
|
28
|
+
}
|
|
29
|
+
function pruneCaches(now) {
|
|
30
|
+
for (const [store, touchedAt] of [...liveCaches]) {
|
|
31
|
+
if (now - touchedAt > CACHE_TTL_MS) {
|
|
32
|
+
store.dropCache();
|
|
33
|
+
liveCaches.delete(store);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
while (liveCaches.size > MAX_CACHED_STORES) {
|
|
37
|
+
const oldest = liveCaches.keys().next().value;
|
|
38
|
+
if (!oldest)
|
|
39
|
+
break;
|
|
40
|
+
oldest.dropCache();
|
|
41
|
+
liveCaches.delete(oldest);
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
// TTL alone only fires on activity, so an idle daemon would hold its last caches
|
|
45
|
+
// forever. A low-frequency sweeper lets a quiet daemon settle back down; unref'd
|
|
46
|
+
// so it never keeps the process alive on its own.
|
|
47
|
+
const sweeper = setInterval(() => pruneCaches(Date.now()), 60_000);
|
|
48
|
+
if (typeof sweeper.unref === "function")
|
|
49
|
+
sweeper.unref();
|
|
50
|
+
export function embeddingCacheStats() {
|
|
51
|
+
return {
|
|
52
|
+
cachedStores: liveCaches.size,
|
|
53
|
+
diskReads,
|
|
54
|
+
reads: diskReads,
|
|
55
|
+
expireOlderThan: (now) => pruneCaches(now)
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
/** Test helper: forget every cached sidecar. */
|
|
59
|
+
export function resetEmbeddingCaches() {
|
|
60
|
+
for (const store of [...liveCaches.keys()])
|
|
61
|
+
store.dropCache();
|
|
62
|
+
liveCaches.clear();
|
|
63
|
+
diskReads = 0;
|
|
64
|
+
}
|
|
4
65
|
/** Real output width per embedding model, learned once per process. */
|
|
5
66
|
const modelDimensions = new Map();
|
|
6
67
|
/** Test helper: forget learned widths, simulating a fresh daemon process. */
|
|
@@ -29,8 +90,11 @@ export class EmbeddingStore {
|
|
|
29
90
|
catch {
|
|
30
91
|
mtimeMs = 0; // missing file → treat as empty, mtime 0
|
|
31
92
|
}
|
|
32
|
-
if (this.cache && this.cache.mtimeMs === mtimeMs)
|
|
93
|
+
if (this.cache && this.cache.mtimeMs === mtimeMs) {
|
|
94
|
+
touchCache(this, Date.now());
|
|
33
95
|
return this.cache.map;
|
|
96
|
+
}
|
|
97
|
+
diskReads += 1;
|
|
34
98
|
const raw = await readFile(this.filePath, "utf8").catch(() => "");
|
|
35
99
|
const map = new Map();
|
|
36
100
|
for (const line of raw.split(/\r?\n/)) {
|
|
@@ -47,6 +111,7 @@ export class EmbeddingStore {
|
|
|
47
111
|
}
|
|
48
112
|
}
|
|
49
113
|
this.cache = { mtimeMs, map };
|
|
114
|
+
touchCache(this, Date.now());
|
|
50
115
|
return map;
|
|
51
116
|
}
|
|
52
117
|
/**
|
|
@@ -63,19 +128,19 @@ export class EmbeddingStore {
|
|
|
63
128
|
const liveIds = new Set(records.map((record) => record.id));
|
|
64
129
|
const pruned = [...existing.keys()].filter((id) => !liveIds.has(id)).length;
|
|
65
130
|
// A stored vector can carry the right model name and hash yet the wrong width —
|
|
66
|
-
// that is what a degraded fallback wrote — and cosineSimilarity
|
|
67
|
-
//
|
|
68
|
-
// Width is part of validity
|
|
69
|
-
//
|
|
70
|
-
//
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
return prior && prior.model === client.model && prior.hash === contentHash(embeddingText(record))
|
|
74
|
-
? prior
|
|
75
|
-
: undefined;
|
|
76
|
-
};
|
|
131
|
+
// that is exactly what a degraded fallback wrote — and cosineSimilarity scores any
|
|
132
|
+
// width mismatch as 0, so those records vanish from semantic recall without ever
|
|
133
|
+
// erroring. Width is part of validity, so we need to know the client's real width.
|
|
134
|
+
//
|
|
135
|
+
// Learning it must not cost a round trip on every sync: the width is cached per
|
|
136
|
+
// model for the process, and only probed when there is nothing to compute (the
|
|
137
|
+
// one case where a fully-poisoned sidecar would otherwise look entirely reusable).
|
|
77
138
|
let expectedDim = modelDimensions.get(client.model) ?? 0;
|
|
78
|
-
|
|
139
|
+
const nothingToRecompute = records.every((record) => {
|
|
140
|
+
const prior = existing.get(record.id);
|
|
141
|
+
return prior && prior.model === client.model && prior.hash === contentHash(embeddingText(record));
|
|
142
|
+
});
|
|
143
|
+
if (expectedDim === 0 && records.length > 0 && nothingToRecompute) {
|
|
79
144
|
try {
|
|
80
145
|
const probe = await client.embed([embeddingText(records[0])]);
|
|
81
146
|
if (!client.degraded && probe[0]?.length) {
|
|
@@ -87,11 +152,15 @@ export class EmbeddingStore {
|
|
|
87
152
|
expectedDim = 0; // cannot probe — fall back to model+hash validity only
|
|
88
153
|
}
|
|
89
154
|
}
|
|
90
|
-
const
|
|
155
|
+
const validDim = (vector) => expectedDim === 0 || vector.length === expectedDim;
|
|
91
156
|
const toCompute = [];
|
|
92
157
|
let reused = 0;
|
|
93
158
|
for (const record of records) {
|
|
94
|
-
|
|
159
|
+
const prior = existing.get(record.id);
|
|
160
|
+
if (prior &&
|
|
161
|
+
prior.model === client.model &&
|
|
162
|
+
prior.hash === contentHash(embeddingText(record)) &&
|
|
163
|
+
validDim(prior.vector)) {
|
|
95
164
|
reused += 1;
|
|
96
165
|
}
|
|
97
166
|
else {
|
|
@@ -100,17 +169,21 @@ export class EmbeddingStore {
|
|
|
100
169
|
}
|
|
101
170
|
const result = new Map();
|
|
102
171
|
for (const record of records) {
|
|
103
|
-
const prior =
|
|
104
|
-
if (
|
|
172
|
+
const prior = existing.get(record.id);
|
|
173
|
+
if (prior &&
|
|
174
|
+
prior.model === client.model &&
|
|
175
|
+
prior.hash === contentHash(embeddingText(record)) &&
|
|
176
|
+
validDim(prior.vector)) {
|
|
105
177
|
result.set(record.id, prior);
|
|
178
|
+
}
|
|
106
179
|
}
|
|
107
180
|
let computed = 0;
|
|
108
181
|
if (toCompute.length > 0) {
|
|
109
182
|
try {
|
|
110
183
|
const vectors = await client.embed(toCompute.map((record) => embeddingText(record)));
|
|
111
184
|
// A degraded run returns local trigram vectors. Serving them for THIS call is
|
|
112
|
-
// graceful degradation; writing them under the primary
|
|
113
|
-
// they would be reused forever as if they were real embeddings.
|
|
185
|
+
// fine (graceful degradation); writing them under the primary's model name is
|
|
186
|
+
// not — they would be reused forever as if they were real embeddings.
|
|
114
187
|
if (client.degraded) {
|
|
115
188
|
const degradedById = new Map();
|
|
116
189
|
for (const [id, stored] of result)
|
|
@@ -148,6 +221,10 @@ export class EmbeddingStore {
|
|
|
148
221
|
vectorById.set(id, stored.vector);
|
|
149
222
|
return { vectorById, computed, reused, pruned };
|
|
150
223
|
}
|
|
224
|
+
/** Release this store's parsed sidecar. Costs one re-read, never correctness. */
|
|
225
|
+
dropCache() {
|
|
226
|
+
this.cache = undefined;
|
|
227
|
+
}
|
|
151
228
|
/** Read vectors without recomputing — used by read-only retrieval paths. */
|
|
152
229
|
async vectorById() {
|
|
153
230
|
const stored = await this.load();
|
|
@@ -170,6 +247,7 @@ export class EmbeddingStore {
|
|
|
170
247
|
await writeFile(tmp, lines.length > 0 ? `${lines.join("\n")}\n` : "", "utf8");
|
|
171
248
|
await rename(tmp, this.filePath);
|
|
172
249
|
this.cache = undefined; // invalidate; next load() re-reads the fresh file
|
|
250
|
+
liveCaches.delete(this);
|
|
173
251
|
}
|
|
174
252
|
}
|
|
175
253
|
/** Embed the record type alongside content so type acts as a soft semantic anchor. */
|
|
@@ -1,9 +1,10 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
|
|
1
2
|
const DEFAULT_MAX_ITEMS = 40;
|
|
2
3
|
const DEFAULT_SNIPPET_CHARS = 240;
|
|
3
4
|
export async function extractDomainEntitiesViaModel(items, options) {
|
|
4
5
|
const out = new Map();
|
|
5
6
|
const { config } = options;
|
|
6
|
-
if (items.length === 0 ||
|
|
7
|
+
if (items.length === 0 || !llmEnabled(config))
|
|
7
8
|
return out;
|
|
8
9
|
const doFetch = options.fetchImpl ?? globalThis.fetch;
|
|
9
10
|
if (!doFetch)
|
|
@@ -19,9 +20,9 @@ export async function extractDomainEntitiesViaModel(items, options) {
|
|
|
19
20
|
'{"n": <snippet number>, "entities": ["..."]}, empty array when a snippet names none. No prose, no fences.';
|
|
20
21
|
const user = `Snippets:\n${numbered}\n\nJSON array:`;
|
|
21
22
|
try {
|
|
22
|
-
const response = await doFetch(
|
|
23
|
+
const response = await doFetch(llmEndpoint(config), {
|
|
23
24
|
method: "POST",
|
|
24
|
-
headers:
|
|
25
|
+
headers: llmHeaders(config),
|
|
25
26
|
body: JSON.stringify({
|
|
26
27
|
model: options.model ?? config.processingModel,
|
|
27
28
|
messages: [
|
|
@@ -1,5 +1,6 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
|
|
1
2
|
export function createGlobalExtractor(config) {
|
|
2
|
-
if (
|
|
3
|
+
if (!llmEnabled(config))
|
|
3
4
|
return null;
|
|
4
5
|
return async (records) => {
|
|
5
6
|
// Send the highest-signal beliefs only — bounds tokens, focuses the model.
|
|
@@ -22,9 +23,9 @@ export function createGlobalExtractor(config) {
|
|
|
22
23
|
"Example DROP (project-internal): 'The daemon exposes a /global/extract endpoint.' " +
|
|
23
24
|
"Rewrite each as one self-contained sentence with zero project context. " +
|
|
24
25
|
"Output ONLY a JSON array of strings — no markdown fences, no prose. If nothing qualifies, return []. Max 8 items.";
|
|
25
|
-
const response = await fetch(
|
|
26
|
+
const response = await fetch(llmEndpoint(config), {
|
|
26
27
|
method: "POST",
|
|
27
|
-
headers:
|
|
28
|
+
headers: llmHeaders(config),
|
|
28
29
|
body: JSON.stringify({
|
|
29
30
|
model: config.processingModel,
|
|
30
31
|
messages: [
|
package/dist/hyde.js
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint } from "./config.js";
|
|
1
2
|
const DEFAULT_MAX_CHARS = 320;
|
|
2
3
|
export async function expandQuery(query, options) {
|
|
3
4
|
const q = (query ?? "").trim();
|
|
4
5
|
if (!q)
|
|
5
6
|
return { expanded: "", hypothetical: "" };
|
|
6
7
|
const { config } = options;
|
|
7
|
-
if (
|
|
8
|
+
if (!llmEnabled(config))
|
|
8
9
|
return { expanded: q, hypothetical: "" };
|
|
9
10
|
const doFetch = options.fetchImpl ?? globalThis.fetch;
|
|
10
11
|
if (!doFetch)
|
|
@@ -16,10 +17,10 @@ export async function expandQuery(query, options) {
|
|
|
16
17
|
"Do not hedge, do not say you lack context, do not ask questions. Output the sentences only.";
|
|
17
18
|
const user = `Question: ${q}\n\nHypothetical answer:`;
|
|
18
19
|
try {
|
|
19
|
-
const response = await doFetch(
|
|
20
|
+
const response = await doFetch(llmEndpoint(config), {
|
|
20
21
|
method: "POST",
|
|
21
22
|
headers: {
|
|
22
|
-
Authorization: `Bearer ${config.openRouterApiKey}`,
|
|
23
|
+
Authorization: `Bearer ${config.llmApiKey ?? config.openRouterApiKey ?? ""}`,
|
|
23
24
|
"Content-Type": "application/json"
|
|
24
25
|
},
|
|
25
26
|
body: JSON.stringify({
|
package/dist/logger.d.ts
CHANGED
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
export interface PeonLoggerOptions {
|
|
2
2
|
logDir?: string;
|
|
3
|
+
/** Rotate the live log once it exceeds this many bytes. */
|
|
4
|
+
maxBytes?: number;
|
|
5
|
+
/** Upper bound on how much of the log tail recent() will read. */
|
|
6
|
+
tailBytes?: number;
|
|
3
7
|
}
|
|
4
8
|
export interface PeonLogEntry {
|
|
5
9
|
id: string;
|
|
@@ -9,9 +13,25 @@ export interface PeonLogEntry {
|
|
|
9
13
|
}
|
|
10
14
|
export declare class PeonLogger {
|
|
11
15
|
private readonly logFile;
|
|
16
|
+
private readonly maxBytes;
|
|
17
|
+
private readonly tailBytes;
|
|
12
18
|
private writeQueue;
|
|
19
|
+
private bytesWritten;
|
|
20
|
+
private sizeKnown;
|
|
13
21
|
constructor(options?: PeonLoggerOptions);
|
|
14
22
|
log(type: string, fields?: Record<string, unknown>): Promise<PeonLogEntry>;
|
|
23
|
+
/**
|
|
24
|
+
* Newest-first entries from the tail of the log. Cost is bounded by
|
|
25
|
+
* `tailBytes`, not by the size of the file, so this stays flat as the log
|
|
26
|
+
* grows. Entries older than the tail window are not visible here — the log
|
|
27
|
+
* file itself (and its rotated siblings) remain the full record.
|
|
28
|
+
*/
|
|
15
29
|
recent(limit?: number): Promise<PeonLogEntry[]>;
|
|
30
|
+
private readTail;
|
|
31
|
+
/**
|
|
32
|
+
* Move the live log aside once it exceeds maxBytes. History is preserved in a
|
|
33
|
+
* timestamped sibling rather than truncated, so nothing is lost.
|
|
34
|
+
*/
|
|
35
|
+
private rotateIfNeeded;
|
|
16
36
|
private enqueueWrite;
|
|
17
37
|
}
|