peon-mem 1.0.6 → 1.0.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/compression.js +4 -3
- package/dist/config.d.ts +13 -0
- package/dist/config.js +26 -0
- package/dist/daemon.js +9 -3
- package/dist/embedding-store.d.ts +12 -0
- package/dist/embedding-store.js +128 -3
- package/dist/embeddings.d.ts +30 -0
- package/dist/embeddings.js +68 -5
- package/dist/entity-extraction.js +4 -3
- package/dist/global-extraction.js +4 -3
- package/dist/hyde.js +4 -3
- package/dist/logger.d.ts +20 -0
- package/dist/logger.js +73 -6
- package/dist/memory-store.d.ts +5 -1
- package/dist/memory-store.js +158 -12
- package/dist/overview.d.ts +18 -6
- package/dist/overview.js +92 -14
- package/dist/processor.d.ts +36 -0
- package/dist/processor.js +85 -2
- package/dist/quality.d.ts +17 -1
- package/dist/quality.js +152 -25
- package/dist/recuration.js +4 -3
- package/dist/reranker.js +4 -3
- package/dist/tools.d.ts +4 -0
- package/dist/tools.js +27 -1
- package/package.json +3 -3
- package/scripts/peon-operations-watch.mjs +80 -0
package/dist/compression.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
|
|
1
2
|
/**
|
|
2
3
|
* Builds the LLM summarizer the brain uses to compress a topic cluster into one
|
|
3
4
|
* gist belief. Kept separate from brain.ts so the curation logic stays pure and
|
|
@@ -5,7 +6,7 @@
|
|
|
5
6
|
* or no key is configured (the brain then runs cost-free, skipping compression).
|
|
6
7
|
*/
|
|
7
8
|
export function createClusterSummarizer(config) {
|
|
8
|
-
if (
|
|
9
|
+
if (!llmEnabled(config))
|
|
9
10
|
return null;
|
|
10
11
|
return async (cluster) => {
|
|
11
12
|
const beliefs = cluster.members.map((m, i) => `${i + 1}. ${m.content}`).join("\n");
|
|
@@ -14,9 +15,9 @@ export function createClusterSummarizer(config) {
|
|
|
14
15
|
"Losing a specific fact is a failure; merge wording, never drop information. Drop only redundancy and filler. " +
|
|
15
16
|
"Output ONLY the summary sentence(s) — no preamble, no markdown, no quotes around it. Max 240 characters.";
|
|
16
17
|
const user = `Topic: ${cluster.entity}\n\nBeliefs to compress:\n${beliefs}\n\nOne compact summary:`;
|
|
17
|
-
const response = await fetch(
|
|
18
|
+
const response = await fetch(llmEndpoint(config), {
|
|
18
19
|
method: "POST",
|
|
19
|
-
headers:
|
|
20
|
+
headers: llmHeaders(config),
|
|
20
21
|
body: JSON.stringify({
|
|
21
22
|
model: config.processingModel,
|
|
22
23
|
messages: [
|
package/dist/config.d.ts
CHANGED
|
@@ -19,4 +19,17 @@ export interface PeonConfig {
|
|
|
19
19
|
type Env = Record<string, string | undefined>;
|
|
20
20
|
export declare function loadPeonConfig(env?: Env): PeonConfig;
|
|
21
21
|
export declare function readEnvFile(startDir?: string): Env;
|
|
22
|
+
/**
|
|
23
|
+
* Is an LLM available for optional AI passes (compression, entity extraction, HyDE,
|
|
24
|
+
* global extraction)?
|
|
25
|
+
*
|
|
26
|
+
* These used to gate on `openRouterApiKey`, which meant a fully-local setup
|
|
27
|
+
* (PEON_PROVIDER=ollama, no OpenRouter key) silently skipped every one of them —
|
|
28
|
+
* "local mode" was not actually local. A local provider needs no key; a hosted one does.
|
|
29
|
+
*/
|
|
30
|
+
export declare function llmEnabled(config: PeonConfig): boolean;
|
|
31
|
+
/** The chat-completions endpoint for the configured provider. */
|
|
32
|
+
export declare function llmEndpoint(config: PeonConfig): string;
|
|
33
|
+
/** Auth + content headers for the configured provider (local providers need no key). */
|
|
34
|
+
export declare function llmHeaders(config: PeonConfig): Record<string, string>;
|
|
22
35
|
export {};
|
package/dist/config.js
CHANGED
|
@@ -97,3 +97,29 @@ function findEnvFile(startDir) {
|
|
|
97
97
|
current = dirname(current);
|
|
98
98
|
}
|
|
99
99
|
}
|
|
100
|
+
/**
|
|
101
|
+
* Is an LLM available for optional AI passes (compression, entity extraction, HyDE,
|
|
102
|
+
* global extraction)?
|
|
103
|
+
*
|
|
104
|
+
* These used to gate on `openRouterApiKey`, which meant a fully-local setup
|
|
105
|
+
* (PEON_PROVIDER=ollama, no OpenRouter key) silently skipped every one of them —
|
|
106
|
+
* "local mode" was not actually local. A local provider needs no key; a hosted one does.
|
|
107
|
+
*/
|
|
108
|
+
export function llmEnabled(config) {
|
|
109
|
+
if (config.aiMode === "off")
|
|
110
|
+
return false;
|
|
111
|
+
if (config.provider === "ollama")
|
|
112
|
+
return true;
|
|
113
|
+
return Boolean(config.llmApiKey ?? config.openRouterApiKey);
|
|
114
|
+
}
|
|
115
|
+
/** The chat-completions endpoint for the configured provider. */
|
|
116
|
+
export function llmEndpoint(config) {
|
|
117
|
+
return `${config.llmBaseUrl.replace(/\/$/, "")}/chat/completions`;
|
|
118
|
+
}
|
|
119
|
+
/** Auth + content headers for the configured provider (local providers need no key). */
|
|
120
|
+
export function llmHeaders(config) {
|
|
121
|
+
return {
|
|
122
|
+
Authorization: `Bearer ${config.llmApiKey ?? config.openRouterApiKey ?? ""}`,
|
|
123
|
+
"Content-Type": "application/json"
|
|
124
|
+
};
|
|
125
|
+
}
|
package/dist/daemon.js
CHANGED
|
@@ -102,9 +102,9 @@ export function canonicalProjectPath(projectPath, home = homedir()) {
|
|
|
102
102
|
/**
|
|
103
103
|
* Reject requests that aren't from a loopback caller — defeats DNS-rebinding and drive-by-localhost
|
|
104
104
|
* attacks where a malicious web page POSTs to the daemon (which would otherwise write/read a brain
|
|
105
|
-
* at an attacker-controlled path). A bad Host header (rebinding)
|
|
106
|
-
* (browser drive-by) is refused. The node hook (no
|
|
107
|
-
* Origin)
|
|
105
|
+
* at an attacker-controlled path). A bad Host header (rebinding), a cross-origin Origin/Referer
|
|
106
|
+
* (browser drive-by) or a browser's Sec-Fetch-Site: cross-site is refused. The node hook (no
|
|
107
|
+
* Origin) and the local monitor UI (loopback Origin, same-origin fetches) all pass.
|
|
108
108
|
*/
|
|
109
109
|
function isLoopbackHostname(hostname) {
|
|
110
110
|
const h = hostname.toLowerCase().replace(/^\[|\]$/g, "");
|
|
@@ -128,6 +128,12 @@ function isLocalRequest(request) {
|
|
|
128
128
|
// A present Host must be loopback (a missing Host — rare, HTTP/1.0 — is allowed; bind is 127.0.0.1).
|
|
129
129
|
if (rawHost !== undefined && !isLoopbackHostname(hostnameFromHostHeader(String(rawHost))))
|
|
130
130
|
return false;
|
|
131
|
+
// Browsers label every request with where it came from. An <img> or no-referrer fetch from
|
|
132
|
+
// another site carries no Origin or Referer, so the loop below cannot see it, yet a GET like
|
|
133
|
+
// /context?projectPath=... still creates a brain at that path. Non-browser callers (the hook,
|
|
134
|
+
// the MCP server, curl) send no Sec-Fetch-Site and are unaffected.
|
|
135
|
+
if (String(request.headers["sec-fetch-site"] ?? "").toLowerCase() === "cross-site")
|
|
136
|
+
return false;
|
|
131
137
|
for (const header of [request.headers.origin, request.headers.referer]) {
|
|
132
138
|
if (!header)
|
|
133
139
|
continue;
|
|
@@ -20,6 +20,16 @@ export interface SyncResult {
|
|
|
20
20
|
reused: number;
|
|
21
21
|
pruned: number;
|
|
22
22
|
}
|
|
23
|
+
export declare function embeddingCacheStats(): {
|
|
24
|
+
cachedStores: number;
|
|
25
|
+
diskReads: number;
|
|
26
|
+
reads: number;
|
|
27
|
+
expireOlderThan: (now: number) => void;
|
|
28
|
+
};
|
|
29
|
+
/** Test helper: forget every cached sidecar. */
|
|
30
|
+
export declare function resetEmbeddingCaches(): void;
|
|
31
|
+
/** Test helper: forget learned widths, simulating a fresh daemon process. */
|
|
32
|
+
export declare function resetEmbeddingDimensionCache(): void;
|
|
23
33
|
export declare class EmbeddingStore {
|
|
24
34
|
private readonly filePath;
|
|
25
35
|
private cache?;
|
|
@@ -33,6 +43,8 @@ export declare class EmbeddingStore {
|
|
|
33
43
|
* empty map rather than throwing (retrieval falls back to lexical-only).
|
|
34
44
|
*/
|
|
35
45
|
sync(records: MemoryRecord[], client: EmbeddingClient | null): Promise<SyncResult>;
|
|
46
|
+
/** Release this store's parsed sidecar. Costs one re-read, never correctness. */
|
|
47
|
+
dropCache(): void;
|
|
36
48
|
/** Read vectors without recomputing — used by read-only retrieval paths. */
|
|
37
49
|
vectorById(): Promise<Map<string, EmbeddingVector>>;
|
|
38
50
|
private persist;
|
package/dist/embedding-store.js
CHANGED
|
@@ -1,6 +1,73 @@
|
|
|
1
1
|
import { mkdir, readFile, rename, stat, writeFile } from "node:fs/promises";
|
|
2
2
|
import { dirname, join } from "node:path";
|
|
3
3
|
import { contentHash } from "./embeddings.js";
|
|
4
|
+
/**
|
|
5
|
+
* Bounded registry of live sidecar caches.
|
|
6
|
+
*
|
|
7
|
+
* The cache below keeps a whole embeddings.jsonl parsed in memory so read-only
|
|
8
|
+
* retrieval doesn't re-parse a multi-MB file on every prompt. That is a real win,
|
|
9
|
+
* but the daemon holds one store per project and nothing evicted: a heap snapshot
|
|
10
|
+
* showed 8 stores pinning 60,288 Float32Array vectors / 338 MB of native backing,
|
|
11
|
+
* with RSS past 1.9 GB — enough GC pressure to peg the CPU and stop the daemon
|
|
12
|
+
* answering. So caches are now LRU-bounded and released once idle; a dropped cache
|
|
13
|
+
* costs one re-read, never correctness.
|
|
14
|
+
*/
|
|
15
|
+
const MAX_CACHED_STORES = Number(process.env.PEON_EMBED_CACHE_STORES) > 0
|
|
16
|
+
? Number(process.env.PEON_EMBED_CACHE_STORES)
|
|
17
|
+
: 2;
|
|
18
|
+
const CACHE_TTL_MS = Number(process.env.PEON_EMBED_CACHE_TTL_MS) > 0
|
|
19
|
+
? Number(process.env.PEON_EMBED_CACHE_TTL_MS)
|
|
20
|
+
: 5 * 60 * 1000;
|
|
21
|
+
/** Insertion order is LRU order: least-recently-used first. */
|
|
22
|
+
const liveCaches = new Map();
|
|
23
|
+
let diskReads = 0;
|
|
24
|
+
function touchCache(store, at) {
|
|
25
|
+
liveCaches.delete(store);
|
|
26
|
+
liveCaches.set(store, at);
|
|
27
|
+
pruneCaches(at);
|
|
28
|
+
}
|
|
29
|
+
function pruneCaches(now) {
|
|
30
|
+
for (const [store, touchedAt] of [...liveCaches]) {
|
|
31
|
+
if (now - touchedAt > CACHE_TTL_MS) {
|
|
32
|
+
store.dropCache();
|
|
33
|
+
liveCaches.delete(store);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
while (liveCaches.size > MAX_CACHED_STORES) {
|
|
37
|
+
const oldest = liveCaches.keys().next().value;
|
|
38
|
+
if (!oldest)
|
|
39
|
+
break;
|
|
40
|
+
oldest.dropCache();
|
|
41
|
+
liveCaches.delete(oldest);
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
// TTL alone only fires on activity, so an idle daemon would hold its last caches
|
|
45
|
+
// forever. A low-frequency sweeper lets a quiet daemon settle back down; unref'd
|
|
46
|
+
// so it never keeps the process alive on its own.
|
|
47
|
+
const sweeper = setInterval(() => pruneCaches(Date.now()), 60_000);
|
|
48
|
+
if (typeof sweeper.unref === "function")
|
|
49
|
+
sweeper.unref();
|
|
50
|
+
export function embeddingCacheStats() {
|
|
51
|
+
return {
|
|
52
|
+
cachedStores: liveCaches.size,
|
|
53
|
+
diskReads,
|
|
54
|
+
reads: diskReads,
|
|
55
|
+
expireOlderThan: (now) => pruneCaches(now)
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
/** Test helper: forget every cached sidecar. */
|
|
59
|
+
export function resetEmbeddingCaches() {
|
|
60
|
+
for (const store of [...liveCaches.keys()])
|
|
61
|
+
store.dropCache();
|
|
62
|
+
liveCaches.clear();
|
|
63
|
+
diskReads = 0;
|
|
64
|
+
}
|
|
65
|
+
/** Real output width per embedding model, learned once per process. */
|
|
66
|
+
const modelDimensions = new Map();
|
|
67
|
+
/** Test helper: forget learned widths, simulating a fresh daemon process. */
|
|
68
|
+
export function resetEmbeddingDimensionCache() {
|
|
69
|
+
modelDimensions.clear();
|
|
70
|
+
}
|
|
4
71
|
export class EmbeddingStore {
|
|
5
72
|
filePath;
|
|
6
73
|
// mtime-keyed cache so the (multi-MB) sidecar isn't re-read+parsed on every prompt's
|
|
@@ -23,8 +90,11 @@ export class EmbeddingStore {
|
|
|
23
90
|
catch {
|
|
24
91
|
mtimeMs = 0; // missing file → treat as empty, mtime 0
|
|
25
92
|
}
|
|
26
|
-
if (this.cache && this.cache.mtimeMs === mtimeMs)
|
|
93
|
+
if (this.cache && this.cache.mtimeMs === mtimeMs) {
|
|
94
|
+
touchCache(this, Date.now());
|
|
27
95
|
return this.cache.map;
|
|
96
|
+
}
|
|
97
|
+
diskReads += 1;
|
|
28
98
|
const raw = await readFile(this.filePath, "utf8").catch(() => "");
|
|
29
99
|
const map = new Map();
|
|
30
100
|
for (const line of raw.split(/\r?\n/)) {
|
|
@@ -41,6 +111,7 @@ export class EmbeddingStore {
|
|
|
41
111
|
}
|
|
42
112
|
}
|
|
43
113
|
this.cache = { mtimeMs, map };
|
|
114
|
+
touchCache(this, Date.now());
|
|
44
115
|
return map;
|
|
45
116
|
}
|
|
46
117
|
/**
|
|
@@ -56,11 +127,40 @@ export class EmbeddingStore {
|
|
|
56
127
|
const existing = await this.load();
|
|
57
128
|
const liveIds = new Set(records.map((record) => record.id));
|
|
58
129
|
const pruned = [...existing.keys()].filter((id) => !liveIds.has(id)).length;
|
|
130
|
+
// A stored vector can carry the right model name and hash yet the wrong width —
|
|
131
|
+
// that is exactly what a degraded fallback wrote — and cosineSimilarity scores any
|
|
132
|
+
// width mismatch as 0, so those records vanish from semantic recall without ever
|
|
133
|
+
// erroring. Width is part of validity, so we need to know the client's real width.
|
|
134
|
+
//
|
|
135
|
+
// Learning it must not cost a round trip on every sync: the width is cached per
|
|
136
|
+
// model for the process, and only probed when there is nothing to compute (the
|
|
137
|
+
// one case where a fully-poisoned sidecar would otherwise look entirely reusable).
|
|
138
|
+
let expectedDim = modelDimensions.get(client.model) ?? 0;
|
|
139
|
+
const nothingToRecompute = records.every((record) => {
|
|
140
|
+
const prior = existing.get(record.id);
|
|
141
|
+
return prior && prior.model === client.model && prior.hash === contentHash(embeddingText(record));
|
|
142
|
+
});
|
|
143
|
+
if (expectedDim === 0 && records.length > 0 && nothingToRecompute) {
|
|
144
|
+
try {
|
|
145
|
+
const probe = await client.embed([embeddingText(records[0])]);
|
|
146
|
+
if (!client.degraded && probe[0]?.length) {
|
|
147
|
+
expectedDim = probe[0].length;
|
|
148
|
+
modelDimensions.set(client.model, expectedDim);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
catch {
|
|
152
|
+
expectedDim = 0; // cannot probe — fall back to model+hash validity only
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
const validDim = (vector) => expectedDim === 0 || vector.length === expectedDim;
|
|
59
156
|
const toCompute = [];
|
|
60
157
|
let reused = 0;
|
|
61
158
|
for (const record of records) {
|
|
62
159
|
const prior = existing.get(record.id);
|
|
63
|
-
if (prior &&
|
|
160
|
+
if (prior &&
|
|
161
|
+
prior.model === client.model &&
|
|
162
|
+
prior.hash === contentHash(embeddingText(record)) &&
|
|
163
|
+
validDim(prior.vector)) {
|
|
64
164
|
reused += 1;
|
|
65
165
|
}
|
|
66
166
|
else {
|
|
@@ -70,7 +170,10 @@ export class EmbeddingStore {
|
|
|
70
170
|
const result = new Map();
|
|
71
171
|
for (const record of records) {
|
|
72
172
|
const prior = existing.get(record.id);
|
|
73
|
-
if (prior &&
|
|
173
|
+
if (prior &&
|
|
174
|
+
prior.model === client.model &&
|
|
175
|
+
prior.hash === contentHash(embeddingText(record)) &&
|
|
176
|
+
validDim(prior.vector)) {
|
|
74
177
|
result.set(record.id, prior);
|
|
75
178
|
}
|
|
76
179
|
}
|
|
@@ -78,6 +181,20 @@ export class EmbeddingStore {
|
|
|
78
181
|
if (toCompute.length > 0) {
|
|
79
182
|
try {
|
|
80
183
|
const vectors = await client.embed(toCompute.map((record) => embeddingText(record)));
|
|
184
|
+
// A degraded run returns local trigram vectors. Serving them for THIS call is
|
|
185
|
+
// fine (graceful degradation); writing them under the primary's model name is
|
|
186
|
+
// not — they would be reused forever as if they were real embeddings.
|
|
187
|
+
if (client.degraded) {
|
|
188
|
+
const degradedById = new Map();
|
|
189
|
+
for (const [id, stored] of result)
|
|
190
|
+
degradedById.set(id, stored.vector);
|
|
191
|
+
toCompute.forEach((record, i) => {
|
|
192
|
+
const vector = vectors[i];
|
|
193
|
+
if (vector)
|
|
194
|
+
degradedById.set(record.id, vector);
|
|
195
|
+
});
|
|
196
|
+
return { vectorById: degradedById, computed: 0, reused, pruned };
|
|
197
|
+
}
|
|
81
198
|
toCompute.forEach((record, i) => {
|
|
82
199
|
result.set(record.id, {
|
|
83
200
|
id: record.id,
|
|
@@ -87,6 +204,9 @@ export class EmbeddingStore {
|
|
|
87
204
|
});
|
|
88
205
|
});
|
|
89
206
|
computed = toCompute.length;
|
|
207
|
+
const width = vectors[0]?.length ?? 0;
|
|
208
|
+
if (width > 0)
|
|
209
|
+
modelDimensions.set(client.model, width);
|
|
90
210
|
}
|
|
91
211
|
catch {
|
|
92
212
|
// On a hard failure, keep whatever we already had and continue lexical-only.
|
|
@@ -101,6 +221,10 @@ export class EmbeddingStore {
|
|
|
101
221
|
vectorById.set(id, stored.vector);
|
|
102
222
|
return { vectorById, computed, reused, pruned };
|
|
103
223
|
}
|
|
224
|
+
/** Release this store's parsed sidecar. Costs one re-read, never correctness. */
|
|
225
|
+
dropCache() {
|
|
226
|
+
this.cache = undefined;
|
|
227
|
+
}
|
|
104
228
|
/** Read vectors without recomputing — used by read-only retrieval paths. */
|
|
105
229
|
async vectorById() {
|
|
106
230
|
const stored = await this.load();
|
|
@@ -123,6 +247,7 @@ export class EmbeddingStore {
|
|
|
123
247
|
await writeFile(tmp, lines.length > 0 ? `${lines.join("\n")}\n` : "", "utf8");
|
|
124
248
|
await rename(tmp, this.filePath);
|
|
125
249
|
this.cache = undefined; // invalidate; next load() re-reads the fresh file
|
|
250
|
+
liveCaches.delete(this);
|
|
126
251
|
}
|
|
127
252
|
}
|
|
128
253
|
/** Embed the record type alongside content so type acts as a soft semantic anchor. */
|
package/dist/embeddings.d.ts
CHANGED
|
@@ -81,6 +81,15 @@ export declare class FallbackEmbeddingClient implements EmbeddingClient {
|
|
|
81
81
|
private readonly fallback;
|
|
82
82
|
private readonly onFallback?;
|
|
83
83
|
readonly model: string;
|
|
84
|
+
/**
|
|
85
|
+
* True when the most recent embed() degraded to the fallback. The vectors it
|
|
86
|
+
* returns are a different model AND a different width, so persisting them under
|
|
87
|
+
* the primary's name makes them indistinguishable from real ones — every later
|
|
88
|
+
* sync then "reuses" trigram vectors as if they were embeddings, and retrieval
|
|
89
|
+
* silently scores them 0 (cosineSimilarity returns 0 on a length mismatch).
|
|
90
|
+
* Callers that persist vectors must check this and skip writing.
|
|
91
|
+
*/
|
|
92
|
+
degraded: boolean;
|
|
84
93
|
constructor(primary: EmbeddingClient, fallback?: EmbeddingClient, onFallback?: ((error: unknown) => void) | undefined);
|
|
85
94
|
embed(texts: string[]): Promise<EmbeddingVector[]>;
|
|
86
95
|
}
|
|
@@ -90,4 +99,25 @@ export interface CreateEmbeddingClientOptions {
|
|
|
90
99
|
onFallback?: (error: unknown) => void;
|
|
91
100
|
}
|
|
92
101
|
/** Build the embedding client implied by config, or null when embeddings are off. */
|
|
102
|
+
/**
|
|
103
|
+
* What was asked for versus what will actually run.
|
|
104
|
+
*
|
|
105
|
+
* Peon degrades to deterministic local trigram embeddings whenever the configured
|
|
106
|
+
* embedder is unavailable. That is deliberate — retrieval keeps working — but it was
|
|
107
|
+
* silent, and a silent downgrade is indistinguishable from working correctly while
|
|
108
|
+
* semantic recall quietly collapses. Two real incidents: an Ollama blip embedding 30k+
|
|
109
|
+
* records with trigram vectors, and a script whose .env was not found resolving to
|
|
110
|
+
* "local" with no warning at all.
|
|
111
|
+
*/
|
|
112
|
+
export interface EmbeddingPlan {
|
|
113
|
+
intended: PeonConfig["embeddingMode"];
|
|
114
|
+
effective: "off" | "local" | "api" | "ollama";
|
|
115
|
+
downgraded: boolean;
|
|
116
|
+
reason?: string;
|
|
117
|
+
}
|
|
118
|
+
/** Only the fields the decision actually depends on, matching the client factory. */
|
|
119
|
+
export type EmbeddingPlanInput = Pick<PeonConfig, "embeddingMode" | "embeddingModel" | "openRouterApiKey" | "provider" | "llmApiKey">;
|
|
120
|
+
export declare function resolveEmbeddingPlan(config: EmbeddingPlanInput): EmbeddingPlan;
|
|
121
|
+
/** Test helper: forget which downgrade warnings have already been emitted. */
|
|
122
|
+
export declare function resetEmbeddingWarnings(): void;
|
|
93
123
|
export declare function createEmbeddingClient(options: CreateEmbeddingClientOptions): EmbeddingClient | null;
|
package/dist/embeddings.js
CHANGED
|
@@ -290,6 +290,15 @@ export class FallbackEmbeddingClient {
|
|
|
290
290
|
fallback;
|
|
291
291
|
onFallback;
|
|
292
292
|
model;
|
|
293
|
+
/**
|
|
294
|
+
* True when the most recent embed() degraded to the fallback. The vectors it
|
|
295
|
+
* returns are a different model AND a different width, so persisting them under
|
|
296
|
+
* the primary's name makes them indistinguishable from real ones — every later
|
|
297
|
+
* sync then "reuses" trigram vectors as if they were embeddings, and retrieval
|
|
298
|
+
* silently scores them 0 (cosineSimilarity returns 0 on a length mismatch).
|
|
299
|
+
* Callers that persist vectors must check this and skip writing.
|
|
300
|
+
*/
|
|
301
|
+
degraded = false;
|
|
293
302
|
constructor(primary, fallback = new LocalEmbeddingClient(), onFallback) {
|
|
294
303
|
this.primary = primary;
|
|
295
304
|
this.fallback = fallback;
|
|
@@ -298,19 +307,73 @@ export class FallbackEmbeddingClient {
|
|
|
298
307
|
}
|
|
299
308
|
async embed(texts) {
|
|
300
309
|
try {
|
|
301
|
-
|
|
310
|
+
const vectors = await this.primary.embed(texts);
|
|
311
|
+
this.degraded = false;
|
|
312
|
+
return vectors;
|
|
302
313
|
}
|
|
303
314
|
catch (error) {
|
|
304
315
|
this.onFallback?.(error);
|
|
316
|
+
this.degraded = true;
|
|
305
317
|
return this.fallback.embed(texts);
|
|
306
318
|
}
|
|
307
319
|
}
|
|
308
320
|
}
|
|
309
|
-
|
|
321
|
+
export function resolveEmbeddingPlan(config) {
|
|
322
|
+
const intended = config.embeddingMode;
|
|
323
|
+
if (intended === "off")
|
|
324
|
+
return { intended, effective: "off", downgraded: false };
|
|
325
|
+
if (intended === "local")
|
|
326
|
+
return { intended, effective: "local", downgraded: false };
|
|
327
|
+
if (intended === "ollama") {
|
|
328
|
+
// Reachability cannot be known at construction time; a dead server surfaces at
|
|
329
|
+
// the first embed() as a degraded run, which the store refuses to persist.
|
|
330
|
+
return { intended, effective: "ollama", downgraded: false };
|
|
331
|
+
}
|
|
332
|
+
const apiKey = config.llmApiKey ?? config.openRouterApiKey;
|
|
333
|
+
if (!apiKey) {
|
|
334
|
+
return { intended, effective: "local", downgraded: true, reason: "no API key/credentials configured" };
|
|
335
|
+
}
|
|
336
|
+
if (!config.embeddingModel) {
|
|
337
|
+
return { intended, effective: "local", downgraded: true, reason: "PEON_EMBEDDING_MODEL is not set" };
|
|
338
|
+
}
|
|
339
|
+
if (config.provider === "anthropic") {
|
|
340
|
+
return { intended, effective: "local", downgraded: true, reason: "Anthropic has no embeddings API" };
|
|
341
|
+
}
|
|
342
|
+
return { intended, effective: "api", downgraded: false };
|
|
343
|
+
}
|
|
344
|
+
/** Warn once per distinct reason, so a long-lived daemon does not spam its log. */
|
|
345
|
+
const warnedDowngrades = new Set();
|
|
346
|
+
function warnOnce(plan) {
|
|
347
|
+
if (!plan.downgraded || !plan.reason)
|
|
348
|
+
return;
|
|
349
|
+
if (warnedDowngrades.has(plan.reason))
|
|
350
|
+
return;
|
|
351
|
+
warnedDowngrades.add(plan.reason);
|
|
352
|
+
console.warn(`[peon] embedding mode "${plan.intended}" is not available (${plan.reason}); ` +
|
|
353
|
+
`falling back to local trigram embeddings. Semantic recall will be much weaker — ` +
|
|
354
|
+
`set PEON_EMBEDDING_MODE=local to silence this, or fix the configuration.`);
|
|
355
|
+
}
|
|
356
|
+
/** Test helper: forget which downgrade warnings have already been emitted. */
|
|
357
|
+
export function resetEmbeddingWarnings() {
|
|
358
|
+
warnedDowngrades.clear();
|
|
359
|
+
}
|
|
310
360
|
export function createEmbeddingClient(options) {
|
|
311
361
|
const { config } = options;
|
|
362
|
+
warnOnce(resolveEmbeddingPlan(config));
|
|
312
363
|
if (config.embeddingMode === "off")
|
|
313
364
|
return null;
|
|
365
|
+
// No caller ever supplied onFallback, so a runtime degrade (embedding server down)
|
|
366
|
+
// was completely silent. Default to warning once per process: the vectors from that
|
|
367
|
+
// run are trigram, not semantic, and the operator needs to know retrieval got worse.
|
|
368
|
+
const onFallback = options.onFallback ??
|
|
369
|
+
((error) => {
|
|
370
|
+
if (warnedDowngrades.has("runtime-fallback"))
|
|
371
|
+
return;
|
|
372
|
+
warnedDowngrades.add("runtime-fallback");
|
|
373
|
+
console.warn(`[peon] embedding request failed (${error instanceof Error ? error.message : String(error)}); ` +
|
|
374
|
+
`falling back to local trigram embeddings for this run. Semantic recall is degraded ` +
|
|
375
|
+
`until the embedding server is reachable again.`);
|
|
376
|
+
});
|
|
314
377
|
if (config.embeddingMode === "ollama") {
|
|
315
378
|
// Local semantic embeddings. Fall back to the API client (if configured) then trigram-local,
|
|
316
379
|
// so a stopped Ollama service degrades instead of breaking retrieval.
|
|
@@ -319,9 +382,9 @@ export function createEmbeddingClient(options) {
|
|
|
319
382
|
baseUrl: config.ollamaBaseUrl
|
|
320
383
|
});
|
|
321
384
|
const fallback = config.openRouterApiKey && config.embeddingModel && config.embeddingModel.includes("/")
|
|
322
|
-
? new FallbackEmbeddingClient(new OpenRouterEmbeddingClient({ apiKey: config.openRouterApiKey, model: config.embeddingModel }), new LocalEmbeddingClient(),
|
|
385
|
+
? new FallbackEmbeddingClient(new OpenRouterEmbeddingClient({ apiKey: config.openRouterApiKey, model: config.embeddingModel }), new LocalEmbeddingClient(), onFallback)
|
|
323
386
|
: new LocalEmbeddingClient();
|
|
324
|
-
return new FallbackEmbeddingClient(ollama, fallback,
|
|
387
|
+
return new FallbackEmbeddingClient(ollama, fallback, onFallback);
|
|
325
388
|
}
|
|
326
389
|
const apiKey = config.llmApiKey ?? config.openRouterApiKey;
|
|
327
390
|
const embeddable = config.provider !== "anthropic"; // Anthropic has no embeddings API — local fallback
|
|
@@ -331,7 +394,7 @@ export function createEmbeddingClient(options) {
|
|
|
331
394
|
model: config.embeddingModel,
|
|
332
395
|
baseUrl: config.llmBaseUrl
|
|
333
396
|
});
|
|
334
|
-
return new FallbackEmbeddingClient(primary, new LocalEmbeddingClient(),
|
|
397
|
+
return new FallbackEmbeddingClient(primary, new LocalEmbeddingClient(), onFallback);
|
|
335
398
|
}
|
|
336
399
|
// Default and "api"-without-credentials both resolve to deterministic local embeddings.
|
|
337
400
|
return new LocalEmbeddingClient();
|
|
@@ -1,9 +1,10 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
|
|
1
2
|
const DEFAULT_MAX_ITEMS = 40;
|
|
2
3
|
const DEFAULT_SNIPPET_CHARS = 240;
|
|
3
4
|
export async function extractDomainEntitiesViaModel(items, options) {
|
|
4
5
|
const out = new Map();
|
|
5
6
|
const { config } = options;
|
|
6
|
-
if (items.length === 0 ||
|
|
7
|
+
if (items.length === 0 || !llmEnabled(config))
|
|
7
8
|
return out;
|
|
8
9
|
const doFetch = options.fetchImpl ?? globalThis.fetch;
|
|
9
10
|
if (!doFetch)
|
|
@@ -19,9 +20,9 @@ export async function extractDomainEntitiesViaModel(items, options) {
|
|
|
19
20
|
'{"n": <snippet number>, "entities": ["..."]}, empty array when a snippet names none. No prose, no fences.';
|
|
20
21
|
const user = `Snippets:\n${numbered}\n\nJSON array:`;
|
|
21
22
|
try {
|
|
22
|
-
const response = await doFetch(
|
|
23
|
+
const response = await doFetch(llmEndpoint(config), {
|
|
23
24
|
method: "POST",
|
|
24
|
-
headers:
|
|
25
|
+
headers: llmHeaders(config),
|
|
25
26
|
body: JSON.stringify({
|
|
26
27
|
model: options.model ?? config.processingModel,
|
|
27
28
|
messages: [
|
|
@@ -1,5 +1,6 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint, llmHeaders } from "./config.js";
|
|
1
2
|
export function createGlobalExtractor(config) {
|
|
2
|
-
if (
|
|
3
|
+
if (!llmEnabled(config))
|
|
3
4
|
return null;
|
|
4
5
|
return async (records) => {
|
|
5
6
|
// Send the highest-signal beliefs only — bounds tokens, focuses the model.
|
|
@@ -22,9 +23,9 @@ export function createGlobalExtractor(config) {
|
|
|
22
23
|
"Example DROP (project-internal): 'The daemon exposes a /global/extract endpoint.' " +
|
|
23
24
|
"Rewrite each as one self-contained sentence with zero project context. " +
|
|
24
25
|
"Output ONLY a JSON array of strings — no markdown fences, no prose. If nothing qualifies, return []. Max 8 items.";
|
|
25
|
-
const response = await fetch(
|
|
26
|
+
const response = await fetch(llmEndpoint(config), {
|
|
26
27
|
method: "POST",
|
|
27
|
-
headers:
|
|
28
|
+
headers: llmHeaders(config),
|
|
28
29
|
body: JSON.stringify({
|
|
29
30
|
model: config.processingModel,
|
|
30
31
|
messages: [
|
package/dist/hyde.js
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
|
+
import { llmEnabled, llmEndpoint } from "./config.js";
|
|
1
2
|
const DEFAULT_MAX_CHARS = 320;
|
|
2
3
|
export async function expandQuery(query, options) {
|
|
3
4
|
const q = (query ?? "").trim();
|
|
4
5
|
if (!q)
|
|
5
6
|
return { expanded: "", hypothetical: "" };
|
|
6
7
|
const { config } = options;
|
|
7
|
-
if (
|
|
8
|
+
if (!llmEnabled(config))
|
|
8
9
|
return { expanded: q, hypothetical: "" };
|
|
9
10
|
const doFetch = options.fetchImpl ?? globalThis.fetch;
|
|
10
11
|
if (!doFetch)
|
|
@@ -16,10 +17,10 @@ export async function expandQuery(query, options) {
|
|
|
16
17
|
"Do not hedge, do not say you lack context, do not ask questions. Output the sentences only.";
|
|
17
18
|
const user = `Question: ${q}\n\nHypothetical answer:`;
|
|
18
19
|
try {
|
|
19
|
-
const response = await doFetch(
|
|
20
|
+
const response = await doFetch(llmEndpoint(config), {
|
|
20
21
|
method: "POST",
|
|
21
22
|
headers: {
|
|
22
|
-
Authorization: `Bearer ${config.openRouterApiKey}`,
|
|
23
|
+
Authorization: `Bearer ${config.llmApiKey ?? config.openRouterApiKey ?? ""}`,
|
|
23
24
|
"Content-Type": "application/json"
|
|
24
25
|
},
|
|
25
26
|
body: JSON.stringify({
|
package/dist/logger.d.ts
CHANGED
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
export interface PeonLoggerOptions {
|
|
2
2
|
logDir?: string;
|
|
3
|
+
/** Rotate the live log once it exceeds this many bytes. */
|
|
4
|
+
maxBytes?: number;
|
|
5
|
+
/** Upper bound on how much of the log tail recent() will read. */
|
|
6
|
+
tailBytes?: number;
|
|
3
7
|
}
|
|
4
8
|
export interface PeonLogEntry {
|
|
5
9
|
id: string;
|
|
@@ -9,9 +13,25 @@ export interface PeonLogEntry {
|
|
|
9
13
|
}
|
|
10
14
|
export declare class PeonLogger {
|
|
11
15
|
private readonly logFile;
|
|
16
|
+
private readonly maxBytes;
|
|
17
|
+
private readonly tailBytes;
|
|
12
18
|
private writeQueue;
|
|
19
|
+
private bytesWritten;
|
|
20
|
+
private sizeKnown;
|
|
13
21
|
constructor(options?: PeonLoggerOptions);
|
|
14
22
|
log(type: string, fields?: Record<string, unknown>): Promise<PeonLogEntry>;
|
|
23
|
+
/**
|
|
24
|
+
* Newest-first entries from the tail of the log. Cost is bounded by
|
|
25
|
+
* `tailBytes`, not by the size of the file, so this stays flat as the log
|
|
26
|
+
* grows. Entries older than the tail window are not visible here — the log
|
|
27
|
+
* file itself (and its rotated siblings) remain the full record.
|
|
28
|
+
*/
|
|
15
29
|
recent(limit?: number): Promise<PeonLogEntry[]>;
|
|
30
|
+
private readTail;
|
|
31
|
+
/**
|
|
32
|
+
* Move the live log aside once it exceeds maxBytes. History is preserved in a
|
|
33
|
+
* timestamped sibling rather than truncated, so nothing is lost.
|
|
34
|
+
*/
|
|
35
|
+
private rotateIfNeeded;
|
|
16
36
|
private enqueueWrite;
|
|
17
37
|
}
|